{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:7UUAU57OSRTVTCC6WW7N2SCUJ7","short_pith_number":"pith:7UUAU57O","schema_version":"1.0","canonical_sha256":"fd280a77ee946759885eb5bedd48544ff716aaa706d4a05da21126de6ba23377","source":{"kind":"arxiv","id":"2503.12559","version":2},"attestation_state":"computed","paper":{"title":"AdaReTaKe: Adaptive Redundancy Reduction to Perceive Longer for Video-language Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.MM"],"primary_cat":"cs.CV","authors_text":"Jianlong Wu, Li Cao, Liqiang Nie, Qingyi Si, Shiyu Zhu, Xiao Wang","submitted_at":"2025-03-16T16:14:52Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have revolutionized video understanding, yet are still limited by context length when processing long videos. Recent methods compress videos by leveraging visual redundancy uniformly, yielding promising results. Nevertheless, our quantitative analysis shows that redundancy varies significantly across time and model layers, necessitating a more flexible compression strategy. We propose AdaReTaKe, a training-free method that flexibly reduces visual redundancy by allocating compression ratios among time and layers with theoretical guarantees. Integrated in"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.12559","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-03-16T16:14:52Z","cross_cats_sorted":["cs.CL","cs.MM"],"title_canon_sha256":"74e29537c015c1d79cf24d96643c34fde6ac4fb3919dfa42e80b1ff43edd25b0","abstract_canon_sha256":"7b4c056feac144bc178fafbd580d9103a8809fc8d4cf99d4961eaa238f2f157c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:17:43.492873Z","signature_b64":"70vXRnkRo0JpJAp+JgVL7xTpYJNvFn4I+w3peOXsdCPAVsqM1n1SzrmHIWiGTHdM8sLmarm7kYZZIz5rs+vICQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fd280a77ee946759885eb5bedd48544ff716aaa706d4a05da21126de6ba23377","last_reissued_at":"2026-07-05T11:17:43.492405Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:17:43.492405Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AdaReTaKe: Adaptive Redundancy Reduction to Perceive Longer for Video-language Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.MM"],"primary_cat":"cs.CV","authors_text":"Jianlong Wu, Li Cao, Liqiang Nie, Qingyi Si, Shiyu Zhu, Xiao Wang","submitted_at":"2025-03-16T16:14:52Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have revolutionized video understanding, yet are still limited by context length when processing long videos. Recent methods compress videos by leveraging visual redundancy uniformly, yielding promising results. Nevertheless, our quantitative analysis shows that redundancy varies significantly across time and model layers, necessitating a more flexible compression strategy. We propose AdaReTaKe, a training-free method that flexibly reduces visual redundancy by allocating compression ratios among time and layers with theoretical guarantees. Integrated in"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.12559","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.12559/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.12559","created_at":"2026-07-05T11:17:43.492465+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.12559v2","created_at":"2026-07-05T11:17:43.492465+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.12559","created_at":"2026-07-05T11:17:43.492465+00:00"},{"alias_kind":"pith_short_12","alias_value":"7UUAU57OSRTV","created_at":"2026-07-05T11:17:43.492465+00:00"},{"alias_kind":"pith_short_16","alias_value":"7UUAU57OSRTVTCC6","created_at":"2026-07-05T11:17:43.492465+00:00"},{"alias_kind":"pith_short_8","alias_value":"7UUAU57O","created_at":"2026-07-05T11:17:43.492465+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2406.08035","citing_title":"LVBench: An Extreme Long Video Understanding Benchmark","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10060","citing_title":"Mosaic: Cross-Modal Clustering for Efficient Video Understanding","ref_index":30,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7UUAU57OSRTVTCC6WW7N2SCUJ7","json":"https://pith.science/pith/7UUAU57OSRTVTCC6WW7N2SCUJ7.json","graph_json":"https://pith.science/api/pith-number/7UUAU57OSRTVTCC6WW7N2SCUJ7/graph.json","events_json":"https://pith.science/api/pith-number/7UUAU57OSRTVTCC6WW7N2SCUJ7/events.json","paper":"https://pith.science/paper/7UUAU57O"},"agent_actions":{"view_html":"https://pith.science/pith/7UUAU57OSRTVTCC6WW7N2SCUJ7","download_json":"https://pith.science/pith/7UUAU57OSRTVTCC6WW7N2SCUJ7.json","view_paper":"https://pith.science/paper/7UUAU57O","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.12559&json=true","fetch_graph":"https://pith.science/api/pith-number/7UUAU57OSRTVTCC6WW7N2SCUJ7/graph.json","fetch_events":"https://pith.science/api/pith-number/7UUAU57OSRTVTCC6WW7N2SCUJ7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7UUAU57OSRTVTCC6WW7N2SCUJ7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7UUAU57OSRTVTCC6WW7N2SCUJ7/action/storage_attestation","attest_author":"https://pith.science/pith/7UUAU57OSRTVTCC6WW7N2SCUJ7/action/author_attestation","sign_citation":"https://pith.science/pith/7UUAU57OSRTVTCC6WW7N2SCUJ7/action/citation_signature","submit_replication":"https://pith.science/pith/7UUAU57OSRTVTCC6WW7N2SCUJ7/action/replication_record"}},"created_at":"2026-07-05T11:17:43.492465+00:00","updated_at":"2026-07-05T11:17:43.492465+00:00"}