{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:TYNO45SLGU4EPSBMS7GVPKTS4M","short_pith_number":"pith:TYNO45SL","schema_version":"1.0","canonical_sha256":"9e1aee764b353847c82c97cd57aa72e328cb5a4fa34e3af7cfca5b1cd38c0abd","source":{"kind":"arxiv","id":"2305.17530","version":1},"attestation_state":"computed","paper":{"title":"PuMer: Pruning and Merging Tokens for Efficient Vision Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Bhargavi Paranjape, Hannaneh Hajishirzi, Qingqing Cao","submitted_at":"2023-05-27T17:16:27Z","abstract_excerpt":"Large-scale vision language (VL) models use Transformers to perform cross-modal interactions between the input text and image. These cross-modal interactions are computationally expensive and memory-intensive due to the quadratic complexity of processing the input image and text. We present PuMer: a token reduction framework that uses text-informed Pruning and modality-aware Merging strategies to progressively reduce the tokens of input image and text, improving model inference speed and reducing memory footprint. PuMer learns to keep salient image tokens related to the input text and merges s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.17530","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-05-27T17:16:27Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"94d3df60c57455a02f8cacf408cba7529606e2f5f5748b6424010ef25748caab","abstract_canon_sha256":"e38c46ba006d7a0b1314c5f94197b7981abb825ec1e62ce61c0e825c6ffc6bb7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:14:45.576372Z","signature_b64":"Lr9UE2L62cN00M5NNXUpSFLVDmvUK7N86eo5iBqzmBjEZf+8dxV09kQW6ixHdIlkKXA+O84lYOxif6O/LSO3DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9e1aee764b353847c82c97cd57aa72e328cb5a4fa34e3af7cfca5b1cd38c0abd","last_reissued_at":"2026-07-05T06:14:45.575967Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:14:45.575967Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PuMer: Pruning and Merging Tokens for Efficient Vision Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Bhargavi Paranjape, Hannaneh Hajishirzi, Qingqing Cao","submitted_at":"2023-05-27T17:16:27Z","abstract_excerpt":"Large-scale vision language (VL) models use Transformers to perform cross-modal interactions between the input text and image. These cross-modal interactions are computationally expensive and memory-intensive due to the quadratic complexity of processing the input image and text. We present PuMer: a token reduction framework that uses text-informed Pruning and modality-aware Merging strategies to progressively reduce the tokens of input image and text, improving model inference speed and reducing memory footprint. PuMer learns to keep salient image tokens related to the input text and merges s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.17530","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.17530/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.17530","created_at":"2026-07-05T06:14:45.576016+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.17530v1","created_at":"2026-07-05T06:14:45.576016+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.17530","created_at":"2026-07-05T06:14:45.576016+00:00"},{"alias_kind":"pith_short_12","alias_value":"TYNO45SLGU4E","created_at":"2026-07-05T06:14:45.576016+00:00"},{"alias_kind":"pith_short_16","alias_value":"TYNO45SLGU4EPSBM","created_at":"2026-07-05T06:14:45.576016+00:00"},{"alias_kind":"pith_short_8","alias_value":"TYNO45SL","created_at":"2026-07-05T06:14:45.576016+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.29350","citing_title":"Fast Enough to Act: Spatio-Temporal Visual Token Merging for Low-Latency Robotic VLMs and VLAs","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00531","citing_title":"State Machine Guided Multi-Relational Synthetic Data from Logs for Anomaly Detection","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17837","citing_title":"Temporal Aware Pruning for Efficient Diffusion-based Video Generation","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17837","citing_title":"Temporal Aware Pruning for Efficient Diffusion-based Video Generation","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2510.18091","citing_title":"Accelerating Vision Transformers with Adaptive Patch Sizes","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2512.19219","citing_title":"Selective LoRA for Visual Tokens and Attention Heads","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2509.02560","citing_title":"FastVGGT: Training-Free Acceleration of Visual Geometry Transformer","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2410.04417","citing_title":"SparseVLM: Visual Token Sparsification for Efficient Vision-Language Model Inference","ref_index":107,"is_internal_anchor":false},{"citing_arxiv_id":"2603.27960","citing_title":"Towards Efficient Large Vision-Language Models: A Comprehensive Survey on Inference Strategies","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TYNO45SLGU4EPSBMS7GVPKTS4M","json":"https://pith.science/pith/TYNO45SLGU4EPSBMS7GVPKTS4M.json","graph_json":"https://pith.science/api/pith-number/TYNO45SLGU4EPSBMS7GVPKTS4M/graph.json","events_json":"https://pith.science/api/pith-number/TYNO45SLGU4EPSBMS7GVPKTS4M/events.json","paper":"https://pith.science/paper/TYNO45SL"},"agent_actions":{"view_html":"https://pith.science/pith/TYNO45SLGU4EPSBMS7GVPKTS4M","download_json":"https://pith.science/pith/TYNO45SLGU4EPSBMS7GVPKTS4M.json","view_paper":"https://pith.science/paper/TYNO45SL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.17530&json=true","fetch_graph":"https://pith.science/api/pith-number/TYNO45SLGU4EPSBMS7GVPKTS4M/graph.json","fetch_events":"https://pith.science/api/pith-number/TYNO45SLGU4EPSBMS7GVPKTS4M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TYNO45SLGU4EPSBMS7GVPKTS4M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TYNO45SLGU4EPSBMS7GVPKTS4M/action/storage_attestation","attest_author":"https://pith.science/pith/TYNO45SLGU4EPSBMS7GVPKTS4M/action/author_attestation","sign_citation":"https://pith.science/pith/TYNO45SLGU4EPSBMS7GVPKTS4M/action/citation_signature","submit_replication":"https://pith.science/pith/TYNO45SLGU4EPSBMS7GVPKTS4M/action/replication_record"}},"created_at":"2026-07-05T06:14:45.576016+00:00","updated_at":"2026-07-05T06:14:45.576016+00:00"}