{"as_of":"2026-08-08T22:21:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:44acaca9e6b29be267e6a995d14ba5e5c17e9708c0261240f3403150e7e3c506","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":41,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":41,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":41,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":41,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T15:31:16.907077Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T09:39:46.765108Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":"2408.14023","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-07-04T09:39:46.765108Z","title":"Video-ccam: Enhancing video-language understanding with causal cross-attention masks for short and long videos","venue":null,"work_id":"2ed72cbe-f6a7-4149-911f-be5ff9cb6d0e","year":2024},"citing_paper":{"arxiv_id":"2406.04264","last_updated":"2025-01-01T15:53:58Z","snapshot_observed_at":"2026-08-03T20:38:36.602554Z","submitted_at":"2024-06-06T17:09:32Z","title":"MLVU: Benchmarking Multi-task Long Video Understanding","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-14T19:55:26.333923Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2406.04264"},"observation_digest":"sha256:807ff89c65b898276ca41ecdcb7fc11e36f5bcabae2bbaae9f06bfc3e0e00f19","observation_id":"e8bbff39-2518-420d-8b3f-584957ae697c","resolution":{"observed_at":"2026-05-14T19:55:26.536953Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":"2408.14023","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-07-04T09:39:46.765108Z","title":"Video-ccam: Enhancing video-language understanding with causal cross-attention masks for short and long videos","venue":null,"work_id":"2ed72cbe-f6a7-4149-911f-be5ff9cb6d0e","year":2024},"citing_paper":{"arxiv_id":"2408.10188","last_updated":"2024-12-13T02:32:06Z","snapshot_observed_at":"2026-08-05T14:57:53.592979Z","submitted_at":"2024-08-19T17:48:08Z","title":"LongVILA: Scaling Long-Context Visual Language Models for Long Videos","version":6},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-17T03:51:25.396887Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2408.10188"},"observation_digest":"sha256:2aa9ca3f4ec9b2088e2ab7e8c433924044943828f03bfebd67e61f9a1d9453bb","observation_id":"8ce1ff7b-9ec1-483c-9524-5056486179ab","resolution":{"observed_at":"2026-05-17T03:51:25.501093Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":"2408.14023","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-07-04T09:39:46.765108Z","title":"Video-ccam: Enhancing video-language understanding with causal cross-attention masks for short and long videos","venue":null,"work_id":"2ed72cbe-f6a7-4149-911f-be5ff9cb6d0e","year":2024},"citing_paper":{"arxiv_id":"2501.00574","last_updated":"2025-07-13T16:21:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-31T18:01:23Z","title":"VideoChat-Flash: Hierarchical Compression for Long-Context Video Modeling","version":4},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-18T04:02:43.261543Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2501.00574"},"observation_digest":"sha256:b4335e7f4d920226e0ce268fb84e8bb9eef5660e11e3cfaddc318753a7d6816d","observation_id":"23d85641-bcfe-4bb7-b319-6c101585e709","resolution":{"observed_at":"2026-05-18T04:02:43.716663Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":"2408.14023","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-07-04T09:39:46.765108Z","title":"Video-ccam: Enhancing video-language understanding with causal cross-attention masks for short and long videos","venue":null,"work_id":"2ed72cbe-f6a7-4149-911f-be5ff9cb6d0e","year":2024},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:91af34f4400f6ddd26a7280ebefbe7d6d5f9e14673460b4b1359e4f3aece7cba","observation_id":"a1e3ce7b-6cf8-4833-9414-91be823de78e","resolution":{"observed_at":"2026-05-17T02:52:20.714773Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":"2408.14023","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-07-04T09:39:46.765108Z","title":"Video-ccam: Enhancing video-language understanding with causal cross-attention masks for short and long videos","venue":null,"work_id":"2ed72cbe-f6a7-4149-911f-be5ff9cb6d0e","year":2024},"citing_paper":{"arxiv_id":"2501.13106","last_updated":"2025-06-03T03:33:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T18:59:46Z","title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","version":4},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-11T01:19:59.603343Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2501.13106"},"observation_digest":"sha256:66b68a3dc1afa77a3f81e63b9036becc6fc0228f8ddd5f9c8368eb9ae4f465aa","observation_id":"fe4d3805-a816-4272-b56a-86137f4f33ec","resolution":{"observed_at":"2026-05-11T01:20:00.255829Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-08-08T15:31:16.907077Z","title":"Video-of-thought: Step-by-step video reasoning from perception to cognition","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.06428","last_updated":"2025-02-11T14:59:25Z","snapshot_observed_at":"2026-08-08T15:25:00.286848Z","submitted_at":"2025-02-10T13:03:05Z","title":"CoS: Chain-of-Shot Prompting for Long Video Understanding","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-08T15:31:16.907077Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2502.06428"},"observation_digest":"sha256:6c93c68eb4456a3d74c78ebf1348a2a71cdde266bbd2c5d7ac15bef90df933b9","observation_id":"e3cfbcff-8b4d-487f-a123-e2ae1da5721e","resolution":{"observed_at":"2026-08-08T15:31:16.907077Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-08-07T15:34:42.981605Z","title":"Video-ccam: Enhancing video-language understanding with causal cross-attention masks for short and long videos.arXiv preprint arXiv:2408.14023, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:42.981605Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:39a6d086de03298ee315b5d5906b6cbe083ff4c17e575bc157bc4ed7cdf478e1","observation_id":"ce17acaf-da91-44b8-a33a-65f9f5f53e0c","resolution":{"observed_at":"2026-08-07T15:34:42.981605Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-08-07T15:20:56.775593Z","title":"Video-ccam: Enhancing video-language understanding with causal cross-attention masks for short and long videos.arXiv preprint arXiv:2408.14023, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15447","last_updated":"2025-05-21T12:29:40Z","snapshot_observed_at":"2026-08-07T15:15:18.405516Z","submitted_at":"2025-05-21T12:29:40Z","title":"ViaRL: Adaptive Temporal Grounding via Visual Iterated Amplification Reinforcement Learning","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T15:20:56.775593Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2505.15447"},"observation_digest":"sha256:ff4760772cf43b339eb787d3b5705c55cfc5c1f70b09943e32db34b3d1e01613","observation_id":"4c265e10-9dcf-4610-9ba4-825985f4bb4d","resolution":{"observed_at":"2026-08-07T15:20:56.775593Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-08-07T11:59:06.421786Z","title":"Video-ccam: Enhancing video-language understanding with causal cross-attention masks for short and long videos, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.00993","last_updated":"2025-06-01T12:49:39Z","snapshot_observed_at":"2026-08-08T18:44:33.509277Z","submitted_at":"2025-06-01T12:49:39Z","title":"FlexSelect: Flexible Token Selection for Efficient Long Video Understanding","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:06.421786Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2506.00993"},"observation_digest":"sha256:e4aa85f75efe74274231067f167a8d09ad7eddb4ef4e3129dab30967dcfa4107","observation_id":"889779c6-2937-4cab-bb15-295b9ff8aeea","resolution":{"observed_at":"2026-08-07T11:59:06.421786Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-08-07T10:28:51.431837Z","title":"Video-ccam: Enhancing video-language understanding with causal cross-attention masks for short and long videos.arXiv preprint arXiv:2408.14023, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05260","last_updated":"2025-06-05T17:21:16Z","snapshot_observed_at":"2026-08-07T10:19:37.470353Z","submitted_at":"2025-06-05T17:21:16Z","title":"LeanPO: Lean Preference Optimization for Likelihood Alignment in Video-LLMs","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:51.431837Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2506.05260"},"observation_digest":"sha256:b14469885c0f0878c1b3f3eea7f2f0187b9258fd3ffa754dd499a29475a9e690","observation_id":"97b1818a-42de-4fdb-820d-985568e79bc4","resolution":{"observed_at":"2026-08-07T10:28:51.431837Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-08-07T10:27:08.491781Z","title":"Video-ccam: Enhancing video-language un- derstanding with causal cross-attention masks for short and long videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05344","last_updated":"2025-07-05T15:40:51Z","snapshot_observed_at":"2026-08-07T10:18:54.361523Z","submitted_at":"2025-06-05T17:59:55Z","title":"SparseMM: Head Sparsity Emerges from Visual Concept Responses in MLLMs","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T10:27:08.491781Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2506.05344"},"observation_digest":"sha256:d0d64de069b9cd22f7c8facf9acf96d679f1b7aeef2424db8d2df697aa987162","observation_id":"7952014f-ff33-40b3-bd24-7d4a7bfe2962","resolution":{"observed_at":"2026-08-07T10:27:08.491781Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-08-06T23:13:08.989110Z","title":"Video-ccam: Enhancing video-language understanding with causal cross-attention masks for short and long videos.arXiv preprint arXiv:2408.14023, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.19225","last_updated":"2025-06-24T01:19:56Z","snapshot_observed_at":"2026-08-07T21:07:46.308380Z","submitted_at":"2025-06-24T01:19:56Z","title":"Video-XL-2: Towards Very Long-Video Understanding Through Task-Aware KV Sparsification","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T23:13:08.989110Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2506.19225"},"observation_digest":"sha256:57fba7f69d7310b7dcd40e53052dd4666f7c577fa7d00aaa66af608cc34f0cf3","observation_id":"d632ca7f-a4c0-4de8-9889-4d3d58abaa42","resolution":{"observed_at":"2026-08-06T23:13:08.989110Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-08-06T22:36:36.868717Z","title":"Video-ccam: Enhancing video-language understanding with causal cross-attention masks for short and long videos.arXiv preprint arXiv:2408.14023, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.21184","last_updated":"2025-06-26T12:43:43Z","snapshot_observed_at":"2026-08-08T13:40:41.778180Z","submitted_at":"2025-06-26T12:43:43Z","title":"Task-Aware KV Compression For Cost-Effective Long Video Understanding","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T22:36:36.868717Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2506.21184"},"observation_digest":"sha256:e946d70d846db99d87c8ccf0a6cc6c14abf0efef9e25557e63145d880e87ecdc","observation_id":"29079ea3-7f5a-45e3-b673-b96d566aed81","resolution":{"observed_at":"2026-08-06T22:36:36.868717Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-08-06T22:15:02.340057Z","title":"Video-ccam: Enhancing video-language un- derstanding with causal cross-attention masks for short and long videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.22139","last_updated":"2025-07-22T07:42:31Z","snapshot_observed_at":"2026-08-06T22:07:43.493361Z","submitted_at":"2025-06-27T11:30:51Z","title":"Q-Frame: Query-aware Frame Selection and Multi-Resolution Adaptation for Video-LLMs","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T22:15:02.340057Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2506.22139"},"observation_digest":"sha256:9966026075b5f1499d6ea5e6f69471aca0f8d26fee7881b9b2b2cf2705c8cb81","observation_id":"55a5ff53-44d3-4273-8475-82950400326c","resolution":{"observed_at":"2026-08-06T22:15:02.340057Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-08-06T20:44:00.121528Z","title":"Video-ccam: Enhancing video-language un- derstanding with causal cross-attention masks for short and long videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.01945","last_updated":"2025-07-09T16:30:21Z","snapshot_observed_at":"2026-08-06T20:37:19.917083Z","submitted_at":"2025-07-02T17:55:50Z","title":"LongAnimation: Long Animation Generation with Dynamic Global-Local Memory","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T20:44:00.121528Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2507.01945"},"observation_digest":"sha256:c9584246c34c55817a65283082c1d313f3bb2768948f35de8997beb71bc82f96","observation_id":"bb93b653-36b1-4e29-a744-433a9651ab04","resolution":{"observed_at":"2026-08-06T20:44:00.121528Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-08-06T04:49:37.951565Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.03039","last_updated":"2025-08-05T03:33:24Z","snapshot_observed_at":"2026-08-07T14:34:47.687957Z","submitted_at":"2025-08-05T03:33:24Z","title":"VideoForest: Person-Anchored Hierarchical Reasoning for Cross-Video Question Answering","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T04:49:37.951565Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2508.03039"},"observation_digest":"sha256:5de3e169218de39db78dd0c66f1b1636c61712717986efbded8439785bd66e8b","observation_id":"d8855dac-cc08-4c80-8e64-ec425fe59203","resolution":{"observed_at":"2026-08-06T04:49:37.951565Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-08-05T15:10:16.727170Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.20478","last_updated":"2026-05-29T09:48:44Z","snapshot_observed_at":"2026-08-05T15:10:01.360889Z","submitted_at":"2025-08-28T06:55:08Z","title":"Video-MTR: Reinforced Multi-Turn Reasoning for Long Video Understanding","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-05T15:10:16.727170Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2508.20478"},"observation_digest":"sha256:e4635c987f0b5c3a5347163777422ecf7af872839639abede21acc2e32a55d5f","observation_id":"444ed85b-6e62-4477-8c1d-a427b44b15fa","resolution":{"observed_at":"2026-08-05T15:10:16.727170Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":"2408.14023","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-07-04T09:39:46.765108Z","title":"Video-ccam: Enhancing video-language understanding with causal cross-attention masks for short and long videos","venue":null,"work_id":"2ed72cbe-f6a7-4149-911f-be5ff9cb6d0e","year":2024},"citing_paper":{"arxiv_id":"2602.20913","last_updated":"2026-04-15T16:09:22Z","snapshot_observed_at":"2026-08-02T12:38:41.181077Z","submitted_at":"2026-02-24T13:49:47Z","title":"LongVideo-R1: Smart Navigation for Low-cost Long Video Understanding","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-15T20:01:31.129959Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2602.20913"},"observation_digest":"sha256:2ce417910089ae57ddbf2fe2ac268b41257e946a69d021f2e2c5c2833f148f8f","observation_id":"f875cfc8-e833-497e-87d3-73a3ddd9fb79","resolution":{"observed_at":"2026-05-15T20:01:33.524954Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":"2408.14023","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-07-04T09:39:46.765108Z","title":"Video-ccam: Enhancing video-language understanding with causal cross-attention masks for short and long videos","venue":null,"work_id":"2ed72cbe-f6a7-4149-911f-be5ff9cb6d0e","year":2024},"citing_paper":{"arxiv_id":"2603.27259","last_updated":"2026-06-18T21:01:40Z","snapshot_observed_at":"2026-08-02T11:52:58.572026Z","submitted_at":"2026-03-28T12:44:19Z","title":"Seeing the Scene Matters: Revealing Forgetting in Video Understanding Models with a Scene-Aware Long-Video Benchmark","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-14T22:05:07.326202Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2603.27259"},"observation_digest":"sha256:8c88405392eb960f27ea43e0a9259df3d2010837ca9275e3b18af43a5af53ae4","observation_id":"ed57979b-9d67-4aca-9efa-d64112ba6835","resolution":{"observed_at":"2026-05-14T22:08:04.419248Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":"2408.14023","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-07-04T09:39:46.765108Z","title":"Video-ccam: Enhancing video-language understanding with causal cross-attention masks for short and long videos","venue":null,"work_id":"2ed72cbe-f6a7-4149-911f-be5ff9cb6d0e","year":2024},"citing_paper":{"arxiv_id":"2604.08077","last_updated":"2026-04-09T10:48:32Z","snapshot_observed_at":"2026-07-06T22:57:13.745326Z","submitted_at":"2026-04-09T10:48:32Z","title":"AdaSpark: Adaptive Sparsity for Efficient Long-Video Understanding","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-10T17:21:47.439019Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2604.08077"},"observation_digest":"sha256:08bfc0790cbb6cdfc362c85c50c7820f60b51312462a2b32ba8edcf2e554cc49","observation_id":"1c61831f-6a19-4f9f-a654-df1feb208d45","resolution":{"observed_at":"2026-05-11T06:56:04.334732Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":"2408.14023","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-07-04T09:39:46.765108Z","title":"Video-ccam: Enhancing video-language understanding with causal cross-attention masks for short and long videos","venue":null,"work_id":"2ed72cbe-f6a7-4149-911f-be5ff9cb6d0e","year":2024},"citing_paper":{"arxiv_id":"2604.14149","last_updated":"2026-04-16T15:48:38Z","snapshot_observed_at":"2026-07-06T23:02:00.082783Z","submitted_at":"2026-04-15T17:59:52Z","title":"One Token per Highly Selective Frame: Towards Extreme Compression for Long Video Understanding","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-10T13:28:58.920442Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2604.14149"},"observation_digest":"sha256:b0e85cd854a382a3fe9b344ee7ac752a3fa120a360a969f6b4f5886507d93020","observation_id":"f2f9886c-ae29-49aa-9743-76d23ee107a3","resolution":{"observed_at":"2026-05-10T13:30:26.648482Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":"2408.14023","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-07-04T09:39:46.765108Z","title":"Video-ccam: Enhancing video-language understanding with causal cross-attention masks for short and long videos","venue":null,"work_id":"2ed72cbe-f6a7-4149-911f-be5ff9cb6d0e","year":2024},"citing_paper":{"arxiv_id":"2605.05848","last_updated":"2026-05-08T13:46:44Z","snapshot_observed_at":"2026-07-06T23:18:22.301347Z","submitted_at":"2026-05-07T08:23:27Z","title":"VideoRouter: Query-Adaptive Dual Routing for Efficient Long-Video Understanding","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-08T14:48:39.444933Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2605.05848"},"observation_digest":"sha256:2e6bfe2bd2cb434f3f330c458fa977326920a8ce2a3bf11236d7bbd57354f7e9","observation_id":"fdd603bd-a208-4c6a-bfda-351152e3382f","resolution":{"observed_at":"2026-05-11T18:41:10.039662Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":"2408.14023","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-07-04T09:39:46.765108Z","title":"Video-ccam: Enhancing video-language understanding with causal cross-attention masks for short and long videos","venue":null,"work_id":"2ed72cbe-f6a7-4149-911f-be5ff9cb6d0e","year":2024},"citing_paper":{"arxiv_id":"2605.05848","last_updated":"2026-05-08T13:46:44Z","snapshot_observed_at":"2026-07-06T23:18:22.301347Z","submitted_at":"2026-05-07T08:23:27Z","title":"VideoRouter: Query-Adaptive Dual Routing for Efficient Long-Video Understanding","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-11T01:57:42.822121Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2605.05848"},"observation_digest":"sha256:cf562cb213d0cfbd1d942ef818dd3ff4b1323da0dee482edcc2f9422865a7d1b","observation_id":"e0758a7f-de1a-4979-95b9-fb764a784d0b","resolution":{"observed_at":"2026-05-11T04:05:56.556078Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":"2408.14023","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-07-04T09:39:46.765108Z","title":"Video-ccam: Enhancing video-language understanding with causal cross-attention masks for short and long videos","venue":null,"work_id":"2ed72cbe-f6a7-4149-911f-be5ff9cb6d0e","year":2024},"citing_paper":{"arxiv_id":"2605.07575","last_updated":"2026-05-11T11:58:53Z","snapshot_observed_at":"2026-07-06T23:19:56.415523Z","submitted_at":"2026-05-08T10:46:10Z","title":"Response-G1: Explicit Scene Graph Modeling for Proactive Streaming Video Understanding","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-05-11T02:30:55.939351Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2605.07575"},"observation_digest":"sha256:1e47ae06bf92a1931a01c1bc64cfe6f41cc65bb7b17dfefc98ee946eb868a87a","observation_id":"f8cf6d9b-4edd-435d-8e9e-65ffefa10bf4","resolution":{"observed_at":"2026-05-11T03:20:56.454668Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":"2408.14023","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-07-04T09:39:46.765108Z","title":"Video-ccam: Enhancing video-language understanding with causal cross-attention masks for short and long videos","venue":null,"work_id":"2ed72cbe-f6a7-4149-911f-be5ff9cb6d0e","year":2024},"citing_paper":{"arxiv_id":"2605.07575","last_updated":"2026-05-11T11:58:53Z","snapshot_observed_at":"2026-07-06T23:19:56.415523Z","submitted_at":"2026-05-08T10:46:10Z","title":"Response-G1: Explicit Scene Graph Modeling for Proactive Streaming Video Understanding","version":2},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-05-12T03:00:34.728880Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2605.07575"},"observation_digest":"sha256:83be7f5f7edefb1513424a83ee49eff2e31a7044fef29f9a332daca0986a46d4","observation_id":"0a27ff8e-79be-4bdd-9476-ea62c810ea37","resolution":{"observed_at":"2026-05-12T03:01:17.707528Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":"2408.14023","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-07-04T09:39:46.765108Z","title":"Video-ccam: Enhancing video-language understanding with causal cross-attention masks for short and long videos","venue":null,"work_id":"2ed72cbe-f6a7-4149-911f-be5ff9cb6d0e","year":2024},"citing_paper":{"arxiv_id":"2605.17921","last_updated":"2026-06-01T06:29:58Z","snapshot_observed_at":"2026-07-06T23:28:50.117638Z","submitted_at":"2026-05-18T06:29:44Z","title":"An Efficient Streaming Video Understanding Framework with Agentic Control","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-05-20T11:30:22.151045Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2605.17921"},"observation_digest":"sha256:8cd1cda48eb5aac90fa251ea494aa3317a645e0a19507450ba83990c1df08ccb","observation_id":"353b1242-ee1c-4f1f-8d1c-6fcd19ab717f","resolution":{"observed_at":"2026-05-20T11:33:14.438058Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":"2408.14023","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-07-04T09:39:46.765108Z","title":"Video-ccam: Enhancing video-language understanding with causal cross-attention masks for short and long videos","venue":null,"work_id":"2ed72cbe-f6a7-4149-911f-be5ff9cb6d0e","year":2024},"citing_paper":{"arxiv_id":"2605.18678","last_updated":"2026-05-20T11:14:23Z","snapshot_observed_at":"2026-08-04T16:01:22.114748Z","submitted_at":"2026-05-18T17:18:24Z","title":"Lance: Unified Multimodal Modeling by Multi-Task Synergy","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-20T11:46:52.658984Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2605.18678"},"observation_digest":"sha256:170094a6434acb81f669e72ac1964936574f2b18ac4a8b824974b1777d58e2ac","observation_id":"4fd44adf-2d77-4b56-b812-061e54bb2403","resolution":{"observed_at":"2026-05-20T11:48:14.924683Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":"2408.14023","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-07-04T09:39:46.765108Z","title":"Video-ccam: Enhancing video-language understanding with causal cross-attention masks for short and long videos","venue":null,"work_id":"2ed72cbe-f6a7-4149-911f-be5ff9cb6d0e","year":2024},"citing_paper":{"arxiv_id":"2605.18678","last_updated":"2026-05-20T11:14:23Z","snapshot_observed_at":"2026-08-04T16:01:22.114748Z","submitted_at":"2026-05-18T17:18:24Z","title":"Lance: Unified Multimodal Modeling by Multi-Task Synergy","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-21T07:56:34.034047Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2605.18678"},"observation_digest":"sha256:2954844c46c2918236326bc98f22018d2098dde2aecf3b18f3b006583c6767c0","observation_id":"60c46325-5641-4837-a59b-d9537c6f88d3","resolution":{"observed_at":"2026-05-21T07:59:50.748177Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":"2408.14023","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-07-04T09:39:46.765108Z","title":"Video-ccam: Enhancing video-language understanding with causal cross-attention masks for short and long videos","venue":null,"work_id":"2ed72cbe-f6a7-4149-911f-be5ff9cb6d0e","year":2024},"citing_paper":{"arxiv_id":"2605.22678","last_updated":"2026-05-21T16:20:31Z","snapshot_observed_at":"2026-08-01T16:56:58.243776Z","submitted_at":"2026-05-21T16:20:31Z","title":"Swift Sampling: Selecting Temporal Surprises via Taylor Series","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-22T05:55:23.479344Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2605.22678"},"observation_digest":"sha256:803256bd173b1b5618e6d168b958a595115ce09dac7a5f2684f4a903020e8c3b","observation_id":"22a9fa5d-4c61-43d1-8f22-ee450655e785","resolution":{"observed_at":"2026-05-22T05:56:08.028902Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":"2408.14023","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-07-04T09:39:46.765108Z","title":"Video-ccam: Enhancing video-language understanding with causal cross-attention masks for short and long videos","venue":null,"work_id":"2ed72cbe-f6a7-4149-911f-be5ff9cb6d0e","year":2024},"citing_paper":{"arxiv_id":"2605.26014","last_updated":"2026-05-25T16:33:00Z","snapshot_observed_at":"2026-07-06T23:35:55.988443Z","submitted_at":"2026-05-25T16:33:00Z","title":"STORM: Internalized Modeling for Spatial-Temporal Reasoning in Video-Language Models","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-29T23:04:21.463842Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2605.26014"},"observation_digest":"sha256:f934f92d5ae0f959e7421d0554d185171c42607e2004de696a9cd4a4a1a2c79d","observation_id":"433ae2e0-4eaa-49c3-a4f7-36f6b24270f2","resolution":{"observed_at":"2026-06-29T23:14:02.202123Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":"2408.14023","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-07-04T09:39:46.765108Z","title":"Video-ccam: Enhancing video-language understanding with causal cross-attention masks for short and long videos","venue":null,"work_id":"2ed72cbe-f6a7-4149-911f-be5ff9cb6d0e","year":2024},"citing_paper":{"arxiv_id":"2605.31069","last_updated":"2026-05-29T09:38:43Z","snapshot_observed_at":"2026-08-07T07:32:37.435310Z","submitted_at":"2026-05-29T09:38:43Z","title":"Towards Effective Long-Video Event Prediction via Multi-Level Event Semantics Mining","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-28T23:16:42.001361Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2605.31069"},"observation_digest":"sha256:caec429f97cbcd3a3b98cd72c01cad7bcb44ab356d07cb783b6bd07a60b00938","observation_id":"172a8c6a-851a-49fb-8c07-0e34cf02566d","resolution":{"observed_at":"2026-06-29T00:02:50.352612Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":"2408.14023","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-07-04T09:39:46.765108Z","title":"Video-ccam: Enhancing video-language understanding with causal cross-attention masks for short and long videos","venue":null,"work_id":"2ed72cbe-f6a7-4149-911f-be5ff9cb6d0e","year":2024},"citing_paper":{"arxiv_id":"2606.06532","last_updated":"2026-06-03T17:47:49Z","snapshot_observed_at":"2026-08-07T12:55:44.925060Z","submitted_at":"2026-06-03T17:47:49Z","title":"GOPAgen: Motion-Aware and Efficient Agentic Long-Video Understanding with Structural Memory and Hierarchical Reasoning","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-28T06:33:32.090913Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2606.06532"},"observation_digest":"sha256:66b7a021bbc92a19d6ad99504abe734ea0421b96f26cf7f65e59a233fefddaab","observation_id":"a60abe6e-cf91-4903-b0b6-abe486c2ae41","resolution":{"observed_at":"2026-07-02T07:56:47.358625Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":"2408.14023","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-07-04T09:39:46.765108Z","title":"Video-ccam: Enhancing video-language understanding with causal cross-attention masks for short and long videos","venue":null,"work_id":"2ed72cbe-f6a7-4149-911f-be5ff9cb6d0e","year":2024},"citing_paper":{"arxiv_id":"2606.12125","last_updated":"2026-06-10T14:19:15Z","snapshot_observed_at":"2026-08-04T05:43:14.707313Z","submitted_at":"2026-06-10T14:19:15Z","title":"Q-Fold: Query-Aware Focus-Context Spatio-Temporal Folding for Long Video Understanding","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-27T10:04:29.739632Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2606.12125"},"observation_digest":"sha256:6d5a72fd7a76b81fb30842d82cf51ca3cfe7e473c6c401f81fa5dc8cb2cfbf3d","observation_id":"b06a113c-fe5c-436d-904e-14032ee29aaa","resolution":{"observed_at":"2026-07-03T10:27:56.056825Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":"2408.14023","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-07-04T09:39:46.765108Z","title":"Video-ccam: Enhancing video-language understanding with causal cross-attention masks for short and long videos","venue":null,"work_id":"2ed72cbe-f6a7-4149-911f-be5ff9cb6d0e","year":2024},"citing_paper":{"arxiv_id":"2606.12195","last_updated":"2026-06-10T15:17:08Z","snapshot_observed_at":"2026-08-01T02:09:41.655807Z","submitted_at":"2026-06-10T15:17:08Z","title":"InternVideo3: Agentify Foundation Models with Multimodal Contextual Reasoning","version":1},"reference_index":249,"source":"arxiv_source","source_observed_at":"2026-06-27T09:48:27.652901Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2606.12195"},"observation_digest":"sha256:eae66e2fd0a7dcf090f8a45d9dd245c0f341a3a1bebd68929bb30c5945c2a4fb","observation_id":"612534ff-5173-43db-934e-1bda7fb6300c","resolution":{"observed_at":"2026-07-03T10:48:02.974132Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":"2408.14023","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-07-04T09:39:46.765108Z","title":"Video-ccam: Enhancing video-language understanding with causal cross-attention masks for short and long videos","venue":null,"work_id":"2ed72cbe-f6a7-4149-911f-be5ff9cb6d0e","year":2024},"citing_paper":{"arxiv_id":"2606.21734","last_updated":"2026-06-19T20:43:49Z","snapshot_observed_at":"2026-08-05T18:05:51.515234Z","submitted_at":"2026-06-19T20:43:49Z","title":"HPP: Hierarchical Programmatic Probing for Long Video Understanding by Decoupling Perception and Reasoning","version":1},"reference_index":140,"source":"arxiv_source","source_observed_at":"2026-06-26T14:19:53.450263Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2606.21734"},"observation_digest":"sha256:3212ea2b3d4f3341a9f0bf806bcee8d417fb1d8d4752b939390423c74350fe45","observation_id":"1cefe5d0-9b92-4e34-afdc-c43f7e417913","resolution":{"observed_at":"2026-07-04T06:39:37.452455Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":"2408.14023","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-07-04T09:39:46.765108Z","title":"Video-ccam: Enhancing video-language understanding with causal cross-attention masks for short and long videos","venue":null,"work_id":"2ed72cbe-f6a7-4149-911f-be5ff9cb6d0e","year":2024},"citing_paper":{"arxiv_id":"2606.22804","last_updated":"2026-07-15T07:05:57Z","snapshot_observed_at":"2026-08-02T18:38:14.382601Z","submitted_at":"2026-06-22T03:29:45Z","title":"CoVStream: Edge-Cloud Collaboration for Understanding of Long Video Streams","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-06-26T09:42:58.861599Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2606.22804"},"observation_digest":"sha256:c900fb361a38d35d9ffd3772ea1c694450be459f9870a2fbddbaae8358211812","observation_id":"dbaf52bc-d04f-4e4f-a079-f604648b05b8","resolution":{"observed_at":"2026-07-04T09:39:46.766518Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":"2408.14023","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-07-04T09:39:46.765108Z","title":"Video-ccam: Enhancing video-language understanding with causal cross-attention masks for short and long videos","venue":null,"work_id":"2ed72cbe-f6a7-4149-911f-be5ff9cb6d0e","year":2024},"citing_paper":{"arxiv_id":"2606.30026","last_updated":"2026-06-29T09:27:02Z","snapshot_observed_at":"2026-08-05T16:53:44.108925Z","submitted_at":"2026-06-29T09:27:02Z","title":"MuseBench: Benchmarking Intent-Level Audiovisual Arts Understanding in MLLMs","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-30T06:25:38.593423Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2606.30026"},"observation_digest":"sha256:9388c7c2669db5365e43996434cd36f7aa120a4a26151dd1794eeba1983a461c","observation_id":"8611e1e8-588d-4426-9146-fe6284f1944d","resolution":{"observed_at":"2026-06-30T06:54:21.335724Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-07-11T08:59:46.244502Z","title":"arXiv preprint arXiv:2408.14023 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.05089","last_updated":"2026-07-06T13:50:15Z","snapshot_observed_at":"2026-07-11T08:59:45.751461Z","submitted_at":"2026-07-06T13:50:15Z","title":"TimeThink: Reasoning with Time for Video LLMs","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-07-11T08:59:46.244502Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2607.05089"},"observation_digest":"sha256:fc4690dc7dedb4ca2eeb704c04481b5c69aae40a137689ec97d2ce4780eab761","observation_id":"440a1036-f026-47e1-8c1a-cf37f656f759","resolution":{"observed_at":"2026-07-11T08:59:46.244502Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-08-01T22:38:42.181503Z","title":"arXiv preprint arXiv:2408.14023 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15689","last_updated":"2026-07-17T07:04:31Z","snapshot_observed_at":"2026-08-06T19:22:15.675454Z","submitted_at":"2026-07-17T07:04:31Z","title":"Efficient Frame Selection for Long Videos at Test Time with Attention-Based MLLM Selectors","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-01T22:38:42.181503Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2607.15689"},"observation_digest":"sha256:d5fb7e114822f4d365669b724b71674b6fde0cbe02e98e4ecac7d1e684ff0ba4","observation_id":"2bbef040-3c0b-47ca-98da-46922382d900","resolution":{"observed_at":"2026-08-01T22:38:42.181503Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-08-02T09:17:00.541193Z","title":"arXiv preprint arXiv:2408.14023 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24794","last_updated":"2026-06-30T10:55:56Z","snapshot_observed_at":"2026-08-07T06:42:16.178257Z","submitted_at":"2026-06-30T10:55:56Z","title":"Reasoning with Memory: A Temporal Granularity-Adaptive Framework for Training-Free Long Video Understanding","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-02T09:17:00.541193Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2607.24794"},"observation_digest":"sha256:cd67e9c79e36c89efc9e208e3315eb2ba8d28a54a588546d6ea6b98f4095a5ef","observation_id":"dc306c73-3c39-4073-9433-30caa848ad63","resolution":{"observed_at":"2026-08-02T09:17:00.541193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-08-07T23:42:35.215228Z","title":"arXiv preprint arXiv:2408.14023 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.05780","last_updated":"2026-08-06T09:15:43Z","snapshot_observed_at":"2026-08-08T22:17:51.983592Z","submitted_at":"2026-08-06T09:15:43Z","title":"Evidence-Driven Dynamic Visual Selector for Efficient Long Video Understanding","version":1},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-08-07T23:42:35.215228Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2608.05780"},"observation_digest":"sha256:0acca469a60b3ab63366193d040c9083759aa662c06e421911e46fd9b628c041","observation_id":"2adaf088-b632-4f60-a978-61943e5b8a5f","resolution":{"observed_at":"2026-08-07T23:42:35.215228Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2408.14023/citation-record","integrity":"/paper/2408.14023/integrity","json":"/paper/2408.14023/citation-record.json","paper":"/paper/2408.14023"},"outbound":[],"paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 41 inbound Pith citation observations for arXiv:2408.14023."}