{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FFW334MDOOQ6T7V7AEIKBZNXUW","short_pith_number":"pith:FFW334MD","schema_version":"1.0","canonical_sha256":"296dbdf18373a1e9febf0110a0e5b7a584a35e89015b1b53e3ec3d7eb64054e0","source":{"kind":"arxiv","id":"2408.14023","version":1},"attestation_state":"computed","paper":{"title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Dian Li, Gang Liu, Hui Wang, Jiajun Fei, Zekun Wang, Zhidong Deng","submitted_at":"2024-08-26T05:27:14Z","abstract_excerpt":"Multi-modal large language models (MLLMs) have demonstrated considerable potential across various downstream tasks that require cross-domain knowledge. MLLMs capable of processing videos, known as Video-MLLMs, have attracted broad interest in video-language understanding. However, videos, especially long videos, contain more visual tokens than images, making them difficult for LLMs to process. Existing works either downsample visual features or extend the LLM context size, risking the loss of high-resolution information or slowing down inference speed. To address these limitations, we apply cr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.14023","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-08-26T05:27:14Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"48629f159ad2e1dc8527f12e636bca9df51167e6077ea5e9f4f1f1959fb633fd","abstract_canon_sha256":"93a7f5d03a9a8fdf49041af1da15703e7527883d4e2840ed05baea3b6dcb68d3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:59:15.064376Z","signature_b64":"ZBcJ+dVhZWpsnEeu+j+paAz65COKEcroxxHg8/cFJT61uFDYK4Qf5UfB5CDPji642gWfbeVLq3GUG4tG1mt3Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"296dbdf18373a1e9febf0110a0e5b7a584a35e89015b1b53e3ec3d7eb64054e0","last_reissued_at":"2026-07-05T08:59:15.063892Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:59:15.063892Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Dian Li, Gang Liu, Hui Wang, Jiajun Fei, Zekun Wang, Zhidong Deng","submitted_at":"2024-08-26T05:27:14Z","abstract_excerpt":"Multi-modal large language models (MLLMs) have demonstrated considerable potential across various downstream tasks that require cross-domain knowledge. MLLMs capable of processing videos, known as Video-MLLMs, have attracted broad interest in video-language understanding. However, videos, especially long videos, contain more visual tokens than images, making them difficult for LLMs to process. Existing works either downsample visual features or extend the LLM context size, risking the loss of high-resolution information or slowing down inference speed. To address these limitations, we apply cr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.14023","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.14023/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.14023","created_at":"2026-07-05T08:59:15.063947+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.14023v1","created_at":"2026-07-05T08:59:15.063947+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.14023","created_at":"2026-07-05T08:59:15.063947+00:00"},{"alias_kind":"pith_short_12","alias_value":"FFW334MDOOQ6","created_at":"2026-07-05T08:59:15.063947+00:00"},{"alias_kind":"pith_short_16","alias_value":"FFW334MDOOQ6T7V7","created_at":"2026-07-05T08:59:15.063947+00:00"},{"alias_kind":"pith_short_8","alias_value":"FFW334MD","created_at":"2026-07-05T08:59:15.063947+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":25,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22804","citing_title":"CoVStream: Edge-Cloud Collaboration for Understanding of Long Video Streams","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21734","citing_title":"HPP: Hierarchical Programmatic Probing for Long Video Understanding by Decoupling Perception and Reasoning","ref_index":140,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12195","citing_title":"InternVideo3: Agentify Foundation Models with Multimodal Contextual Reasoning","ref_index":249,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12125","citing_title":"Q-Fold: Query-Aware Focus-Context Spatio-Temporal Folding for Long Video Understanding","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06532","citing_title":"GOPAgen: Motion-Aware and Efficient Agentic Long-Video Understanding with Structural Memory and Hierarchical Reasoning","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30026","citing_title":"MuseBench: Benchmarking Intent-Level Audiovisual Arts Understanding in MLLMs","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26014","citing_title":"STORM: Internalized Modeling for Spatial-Temporal Reasoning in Video-Language Models","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31069","citing_title":"Towards Effective Long-Video Event Prediction via Multi-Level Event Semantics Mining","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22678","citing_title":"Swift Sampling: Selecting Temporal Surprises via Taylor Series","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18678","citing_title":"Lance: Unified Multimodal Modeling by Multi-Task Synergy","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18678","citing_title":"Lance: Unified Multimodal Modeling by Multi-Task Synergy","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17921","citing_title":"An Efficient Streaming Video Understanding Framework with Agentic Control","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2501.00574","citing_title":"VideoChat-Flash: Hierarchical Compression for Long-Context Video Modeling","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2408.10188","citing_title":"LongVILA: Scaling Long-Context Visual Language Models for Long Videos","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2501.12386","citing_title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2602.20913","citing_title":"LongVideo-R1: Smart Navigation for Low-cost Long Video Understanding","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2603.27259","citing_title":"Seeing the Scene Matters: Revealing Forgetting in Video Understanding Models with a Scene-Aware Long-Video Benchmark","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2406.04264","citing_title":"MLVU: Benchmarking Multi-task Long Video Understanding","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07575","citing_title":"Response-G1: Explicit Scene Graph Modeling for Proactive Streaming Video Understanding","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05848","citing_title":"VideoRouter: Query-Adaptive Dual Routing for Efficient Long-Video Understanding","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08077","citing_title":"AdaSpark: Adaptive Sparsity for Efficient Long-Video Understanding","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05848","citing_title":"VideoRouter: Query-Adaptive Dual Routing for Efficient Long-Video Understanding","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07575","citing_title":"Response-G1: Explicit Scene Graph Modeling for Proactive Streaming Video Understanding","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2501.13106","citing_title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14149","citing_title":"One Token per Highly Selective Frame: Towards Extreme Compression for Long Video Understanding","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FFW334MDOOQ6T7V7AEIKBZNXUW","json":"https://pith.science/pith/FFW334MDOOQ6T7V7AEIKBZNXUW.json","graph_json":"https://pith.science/api/pith-number/FFW334MDOOQ6T7V7AEIKBZNXUW/graph.json","events_json":"https://pith.science/api/pith-number/FFW334MDOOQ6T7V7AEIKBZNXUW/events.json","paper":"https://pith.science/paper/FFW334MD"},"agent_actions":{"view_html":"https://pith.science/pith/FFW334MDOOQ6T7V7AEIKBZNXUW","download_json":"https://pith.science/pith/FFW334MDOOQ6T7V7AEIKBZNXUW.json","view_paper":"https://pith.science/paper/FFW334MD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.14023&json=true","fetch_graph":"https://pith.science/api/pith-number/FFW334MDOOQ6T7V7AEIKBZNXUW/graph.json","fetch_events":"https://pith.science/api/pith-number/FFW334MDOOQ6T7V7AEIKBZNXUW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FFW334MDOOQ6T7V7AEIKBZNXUW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FFW334MDOOQ6T7V7AEIKBZNXUW/action/storage_attestation","attest_author":"https://pith.science/pith/FFW334MDOOQ6T7V7AEIKBZNXUW/action/author_attestation","sign_citation":"https://pith.science/pith/FFW334MDOOQ6T7V7AEIKBZNXUW/action/citation_signature","submit_replication":"https://pith.science/pith/FFW334MDOOQ6T7V7AEIKBZNXUW/action/replication_record"}},"created_at":"2026-07-05T08:59:15.063947+00:00","updated_at":"2026-07-05T08:59:15.063947+00:00"}