{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JHNHKDVJZFG7NFMYKK7SUFCN3W","short_pith_number":"pith:JHNHKDVJ","schema_version":"1.0","canonical_sha256":"49da750ea9c94df6959852bf2a144dddaace332cf289da90d335a8a9a3222964","source":{"kind":"arxiv","id":"2412.20504","version":5},"attestation_state":"computed","paper":{"title":"ReTaKe: Reducing Temporal and Knowledge Redundancy for Long Video Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.MM"],"primary_cat":"cs.CV","authors_text":"Jianlong Wu, Li Cao, Liqiang Nie, Qingyi Si, Shiyu Zhu, Xiao Wang","submitted_at":"2024-12-29T15:42:24Z","abstract_excerpt":"Video Large Language Models (VideoLLMs) have made significant strides in video understanding but struggle with long videos due to the limitations of their backbone LLMs. Existing solutions rely on length extrapolation, which is memory-constrained, or visual token compression, which primarily leverages low-level temporal redundancy while overlooking the more effective high-level knowledge redundancy. To address this, we propose $\\textbf{ReTaKe}$, a training-free method with two novel modules DPSelect and PivotKV, to jointly reduce both temporal visual redundancy and knowledge redundancy for vid"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.20504","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-12-29T15:42:24Z","cross_cats_sorted":["cs.CL","cs.MM"],"title_canon_sha256":"c7816c06b96b656e0b1249abd20e9011520dd1c159ef1376fb38917861f42cb0","abstract_canon_sha256":"1537116914e3b2a0882f99f96aa38b76f40c542a67acc9b370c4f063ffea7977"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:37:41.530996Z","signature_b64":"0AQWK7jfRmCtI6g04nZHJBa8YmOOhN4aOIwfrSeAIfN/UplUw8k0S2j58Zp5bMRDec21ZNReBlRBdFcb+uPEAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"49da750ea9c94df6959852bf2a144dddaace332cf289da90d335a8a9a3222964","last_reissued_at":"2026-07-05T10:37:41.529977Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:37:41.529977Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ReTaKe: Reducing Temporal and Knowledge Redundancy for Long Video Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.MM"],"primary_cat":"cs.CV","authors_text":"Jianlong Wu, Li Cao, Liqiang Nie, Qingyi Si, Shiyu Zhu, Xiao Wang","submitted_at":"2024-12-29T15:42:24Z","abstract_excerpt":"Video Large Language Models (VideoLLMs) have made significant strides in video understanding but struggle with long videos due to the limitations of their backbone LLMs. Existing solutions rely on length extrapolation, which is memory-constrained, or visual token compression, which primarily leverages low-level temporal redundancy while overlooking the more effective high-level knowledge redundancy. To address this, we propose $\\textbf{ReTaKe}$, a training-free method with two novel modules DPSelect and PivotKV, to jointly reduce both temporal visual redundancy and knowledge redundancy for vid"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.20504","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.20504/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.20504","created_at":"2026-07-05T10:37:41.530075+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.20504v5","created_at":"2026-07-05T10:37:41.530075+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.20504","created_at":"2026-07-05T10:37:41.530075+00:00"},{"alias_kind":"pith_short_12","alias_value":"JHNHKDVJZFG7","created_at":"2026-07-05T10:37:41.530075+00:00"},{"alias_kind":"pith_short_16","alias_value":"JHNHKDVJZFG7NFMY","created_at":"2026-07-05T10:37:41.530075+00:00"},{"alias_kind":"pith_short_8","alias_value":"JHNHKDVJ","created_at":"2026-07-05T10:37:41.530075+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2602.20913","citing_title":"LongVideo-R1: Smart Navigation for Low-cost Long Video Understanding","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2602.20981","citing_title":"Echoes Over Time: Unlocking Length Generalization in Video-to-Audio Generation Models","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JHNHKDVJZFG7NFMYKK7SUFCN3W","json":"https://pith.science/pith/JHNHKDVJZFG7NFMYKK7SUFCN3W.json","graph_json":"https://pith.science/api/pith-number/JHNHKDVJZFG7NFMYKK7SUFCN3W/graph.json","events_json":"https://pith.science/api/pith-number/JHNHKDVJZFG7NFMYKK7SUFCN3W/events.json","paper":"https://pith.science/paper/JHNHKDVJ"},"agent_actions":{"view_html":"https://pith.science/pith/JHNHKDVJZFG7NFMYKK7SUFCN3W","download_json":"https://pith.science/pith/JHNHKDVJZFG7NFMYKK7SUFCN3W.json","view_paper":"https://pith.science/paper/JHNHKDVJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.20504&json=true","fetch_graph":"https://pith.science/api/pith-number/JHNHKDVJZFG7NFMYKK7SUFCN3W/graph.json","fetch_events":"https://pith.science/api/pith-number/JHNHKDVJZFG7NFMYKK7SUFCN3W/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JHNHKDVJZFG7NFMYKK7SUFCN3W/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JHNHKDVJZFG7NFMYKK7SUFCN3W/action/storage_attestation","attest_author":"https://pith.science/pith/JHNHKDVJZFG7NFMYKK7SUFCN3W/action/author_attestation","sign_citation":"https://pith.science/pith/JHNHKDVJZFG7NFMYKK7SUFCN3W/action/citation_signature","submit_replication":"https://pith.science/pith/JHNHKDVJZFG7NFMYKK7SUFCN3W/action/replication_record"}},"created_at":"2026-07-05T10:37:41.530075+00:00","updated_at":"2026-07-05T10:37:41.530075+00:00"}