{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:E3QNJ6T6ZW6WPHNFO2QH3RK7BY","short_pith_number":"pith:E3QNJ6T6","schema_version":"1.0","canonical_sha256":"26e0d4fa7ecdbd679da576a07dc55f0e38864acc64038d5a788a4ef3a5ebd122","source":{"kind":"arxiv","id":"2411.15024","version":3},"attestation_state":"computed","paper":{"title":"DyCoke: Dynamic Compression of Tokens for Fast Video Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Can Qin, Haoxuan You, Huan Wang, Keda Tao, Yang Sui","submitted_at":"2024-11-22T15:55:19Z","abstract_excerpt":"Video large language models (VLLMs) have significantly advanced recently in processing complex video content, yet their inference efficiency remains constrained because of the high computational cost stemming from the thousands of visual tokens generated from the video inputs. We empirically observe that, unlike single image inputs, VLLMs typically attend visual tokens from different frames at different decoding iterations, making a one-shot pruning strategy prone to removing important tokens by mistake. Motivated by this, we present DyCoke, a training-free token compression method to optimize"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.15024","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-11-22T15:55:19Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"81880d855dd38a3b0795aca9e73e91d9ea78acf008b6aaae6354d7f38cc0b07c","abstract_canon_sha256":"ca5e8f4a3e449cf7c036accb4413d05dac648da1c7544c05c81412e51f48f439"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:40:59.354988Z","signature_b64":"C0nEzGtD6m5BV39Z+3SO4E5frUspmVh+P0GuemMT4dbc3Bq0Md+Fntd7tI9YyVAFqGWwPYOFf7zQfa+kr2xECg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"26e0d4fa7ecdbd679da576a07dc55f0e38864acc64038d5a788a4ef3a5ebd122","last_reissued_at":"2026-07-05T10:40:59.354449Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:40:59.354449Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DyCoke: Dynamic Compression of Tokens for Fast Video Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Can Qin, Haoxuan You, Huan Wang, Keda Tao, Yang Sui","submitted_at":"2024-11-22T15:55:19Z","abstract_excerpt":"Video large language models (VLLMs) have significantly advanced recently in processing complex video content, yet their inference efficiency remains constrained because of the high computational cost stemming from the thousands of visual tokens generated from the video inputs. We empirically observe that, unlike single image inputs, VLLMs typically attend visual tokens from different frames at different decoding iterations, making a one-shot pruning strategy prone to removing important tokens by mistake. Motivated by this, we present DyCoke, a training-free token compression method to optimize"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.15024","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.15024/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.15024","created_at":"2026-07-05T10:40:59.354507+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.15024v3","created_at":"2026-07-05T10:40:59.354507+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.15024","created_at":"2026-07-05T10:40:59.354507+00:00"},{"alias_kind":"pith_short_12","alias_value":"E3QNJ6T6ZW6W","created_at":"2026-07-05T10:40:59.354507+00:00"},{"alias_kind":"pith_short_16","alias_value":"E3QNJ6T6ZW6WPHNF","created_at":"2026-07-05T10:40:59.354507+00:00"},{"alias_kind":"pith_short_8","alias_value":"E3QNJ6T6","created_at":"2026-07-05T10:40:59.354507+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2505.15269","citing_title":"LiveVLM: Efficient Online Video Understanding via Streaming-Oriented KV Cache and Retrieval","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2507.21420","citing_title":"ReGATE: Learning Faster and Better with Fewer Tokens in MLLMs","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2601.14724","citing_title":"HERMES: KV Cache as Hierarchical Memory for Efficient Streaming Video Understanding","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12358","citing_title":"Why and When Visual Token Pruning Fails? A Study on Relevant Visual Information Shift in MLLMs Decoding","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07355","citing_title":"TTF: Temporal Token Fusion for Efficient Video-Language Model","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/E3QNJ6T6ZW6WPHNFO2QH3RK7BY","json":"https://pith.science/pith/E3QNJ6T6ZW6WPHNFO2QH3RK7BY.json","graph_json":"https://pith.science/api/pith-number/E3QNJ6T6ZW6WPHNFO2QH3RK7BY/graph.json","events_json":"https://pith.science/api/pith-number/E3QNJ6T6ZW6WPHNFO2QH3RK7BY/events.json","paper":"https://pith.science/paper/E3QNJ6T6"},"agent_actions":{"view_html":"https://pith.science/pith/E3QNJ6T6ZW6WPHNFO2QH3RK7BY","download_json":"https://pith.science/pith/E3QNJ6T6ZW6WPHNFO2QH3RK7BY.json","view_paper":"https://pith.science/paper/E3QNJ6T6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.15024&json=true","fetch_graph":"https://pith.science/api/pith-number/E3QNJ6T6ZW6WPHNFO2QH3RK7BY/graph.json","fetch_events":"https://pith.science/api/pith-number/E3QNJ6T6ZW6WPHNFO2QH3RK7BY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/E3QNJ6T6ZW6WPHNFO2QH3RK7BY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/E3QNJ6T6ZW6WPHNFO2QH3RK7BY/action/storage_attestation","attest_author":"https://pith.science/pith/E3QNJ6T6ZW6WPHNFO2QH3RK7BY/action/author_attestation","sign_citation":"https://pith.science/pith/E3QNJ6T6ZW6WPHNFO2QH3RK7BY/action/citation_signature","submit_replication":"https://pith.science/pith/E3QNJ6T6ZW6WPHNFO2QH3RK7BY/action/replication_record"}},"created_at":"2026-07-05T10:40:59.354507+00:00","updated_at":"2026-07-05T10:40:59.354507+00:00"}