{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PF5HXGXMMTHSIOUC6SFWFSF3AG","short_pith_number":"pith:PF5HXGXM","schema_version":"1.0","canonical_sha256":"797a7b9aec64cf243a82f48b62c8bb019d92c0a8a97ed35f8fa5d3cc94e67e28","source":{"kind":"arxiv","id":"2412.16117","version":1},"attestation_state":"computed","paper":{"title":"PruneVid: Visual Token Pruning for Efficient Video Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hao Zhou, Kai Han, Xiaohu Huang","submitted_at":"2024-12-20T18:01:58Z","abstract_excerpt":"In this paper, we introduce PruneVid, a visual token pruning method designed to enhance the efficiency of multi-modal video understanding. Large Language Models (LLMs) have shown promising performance in video tasks due to their extended capabilities in comprehending visual modalities. However, the substantial redundancy in video data presents significant computational challenges for LLMs. To address this issue, we introduce a training-free method that 1) minimizes video redundancy by merging spatial-temporal tokens, and 2) leverages LLMs' reasoning capabilities to selectively prune visual fea"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.16117","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-12-20T18:01:58Z","cross_cats_sorted":[],"title_canon_sha256":"0c6b50853edfe8c2fdadfefcd6e6ccea0d130da1d93e963d59ebedd27f6f57a9","abstract_canon_sha256":"b8599910403b03884b7b68f19b1d7e0c8367e7d7f94695ddf1e9d858deb715a0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:52:35.242180Z","signature_b64":"DOfbWwPhin3n0M0xC85aY/n5TkOMq5aKdehANJtQ8sxHSQ15DyuDggYNxp+9/f5dJ1Tf960b1DjhE8ukPrXSAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"797a7b9aec64cf243a82f48b62c8bb019d92c0a8a97ed35f8fa5d3cc94e67e28","last_reissued_at":"2026-07-05T09:52:35.241719Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:52:35.241719Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PruneVid: Visual Token Pruning for Efficient Video Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hao Zhou, Kai Han, Xiaohu Huang","submitted_at":"2024-12-20T18:01:58Z","abstract_excerpt":"In this paper, we introduce PruneVid, a visual token pruning method designed to enhance the efficiency of multi-modal video understanding. Large Language Models (LLMs) have shown promising performance in video tasks due to their extended capabilities in comprehending visual modalities. However, the substantial redundancy in video data presents significant computational challenges for LLMs. To address this issue, we introduce a training-free method that 1) minimizes video redundancy by merging spatial-temporal tokens, and 2) leverages LLMs' reasoning capabilities to selectively prune visual fea"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.16117","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.16117/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.16117","created_at":"2026-07-05T09:52:35.241778+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.16117v1","created_at":"2026-07-05T09:52:35.241778+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.16117","created_at":"2026-07-05T09:52:35.241778+00:00"},{"alias_kind":"pith_short_12","alias_value":"PF5HXGXMMTHS","created_at":"2026-07-05T09:52:35.241778+00:00"},{"alias_kind":"pith_short_16","alias_value":"PF5HXGXMMTHSIOUC","created_at":"2026-07-05T09:52:35.241778+00:00"},{"alias_kind":"pith_short_8","alias_value":"PF5HXGXM","created_at":"2026-07-05T09:52:35.241778+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2602.22779","citing_title":"TrajTok: Learning Trajectory Tokens enables better Video Understanding","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2603.01400","citing_title":"Token Reduction via Local and Global Contexts Optimization for Efficient Video Large Language Models","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11627","citing_title":"POINTS-Long: Adaptive Dual-Mode Visual Reasoning in MLLMs","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PF5HXGXMMTHSIOUC6SFWFSF3AG","json":"https://pith.science/pith/PF5HXGXMMTHSIOUC6SFWFSF3AG.json","graph_json":"https://pith.science/api/pith-number/PF5HXGXMMTHSIOUC6SFWFSF3AG/graph.json","events_json":"https://pith.science/api/pith-number/PF5HXGXMMTHSIOUC6SFWFSF3AG/events.json","paper":"https://pith.science/paper/PF5HXGXM"},"agent_actions":{"view_html":"https://pith.science/pith/PF5HXGXMMTHSIOUC6SFWFSF3AG","download_json":"https://pith.science/pith/PF5HXGXMMTHSIOUC6SFWFSF3AG.json","view_paper":"https://pith.science/paper/PF5HXGXM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.16117&json=true","fetch_graph":"https://pith.science/api/pith-number/PF5HXGXMMTHSIOUC6SFWFSF3AG/graph.json","fetch_events":"https://pith.science/api/pith-number/PF5HXGXMMTHSIOUC6SFWFSF3AG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PF5HXGXMMTHSIOUC6SFWFSF3AG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PF5HXGXMMTHSIOUC6SFWFSF3AG/action/storage_attestation","attest_author":"https://pith.science/pith/PF5HXGXMMTHSIOUC6SFWFSF3AG/action/author_attestation","sign_citation":"https://pith.science/pith/PF5HXGXMMTHSIOUC6SFWFSF3AG/action/citation_signature","submit_replication":"https://pith.science/pith/PF5HXGXMMTHSIOUC6SFWFSF3AG/action/replication_record"}},"created_at":"2026-07-05T09:52:35.241778+00:00","updated_at":"2026-07-05T09:52:35.241778+00:00"}