{"as_of":"2026-08-08T08:48:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:cea08ef5e9d2d0646cef01a4f29de4a2e0c9dba0111dfe734b02a5be1877043a","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":20,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":20,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":20,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":20,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T21:57:23.556854Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-06-28T23:12:46.636950Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2406.14515","last_updated":"2024-10-30T13:38:10Z","snapshot_observed_at":"2026-07-06T18:34:24.078145Z","submitted_at":"2024-06-20T17:26:01Z","title":"MMBench-Video: A Long-Form Multi-Shot Benchmark for Holistic Video Understanding","version":3},"cited_work":{"arxiv_id":"2406.14515","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.14515","snapshot_observed_at":"2026-06-28T23:12:46.636950Z","title":"Mmbench-video: A long-form multi-shot benchmark for holistic video under- standing","venue":null,"work_id":"0108342f-f183-4a3e-930e-337bd83a6601","year":2024},"citing_paper":{"arxiv_id":"2406.07476","last_updated":"2024-10-30T06:49:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-11T17:22:23Z","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-11T02:44:53.284345Z"},"links":{"cited_paper":"/paper/2406.14515","citing_paper":"/paper/2406.07476"},"observation_digest":"sha256:e69e8d996fdb4654a8766a7326ae49ccab78777a18493880e9c389d3d7d074c1","observation_id":"38f029b6-d9a7-4cfe-a8a0-5b6ca001afd8","resolution":{"observed_at":"2026-05-11T02:44:53.405489Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14515","last_updated":"2024-10-30T13:38:10Z","snapshot_observed_at":"2026-07-06T18:34:24.078145Z","submitted_at":"2024-06-20T17:26:01Z","title":"MMBench-Video: A Long-Form Multi-Shot Benchmark for Holistic Video Understanding","version":3},"cited_work":{"arxiv_id":"2406.14515","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.14515","snapshot_observed_at":"2026-06-28T23:12:46.636950Z","title":"Mmbench-video: A long-form multi-shot benchmark for holistic video under- standing","venue":null,"work_id":"0108342f-f183-4a3e-930e-337bd83a6601","year":2024},"citing_paper":{"arxiv_id":"2407.03320","last_updated":"2024-07-03T17:59:21Z","snapshot_observed_at":"2026-08-04T22:09:42.241578Z","submitted_at":"2024-07-03T17:59:21Z","title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-17T10:46:28.447347Z"},"links":{"cited_paper":"/paper/2406.14515","citing_paper":"/paper/2407.03320"},"observation_digest":"sha256:1d66c28514635d18e578f42bcf2d3b3dd020365e98501ca82b1be327a76c31e4","observation_id":"b8e1403e-d56f-470d-9a35-365fd639fabf","resolution":{"observed_at":"2026-05-17T10:46:28.609435Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14515","last_updated":"2024-10-30T13:38:10Z","snapshot_observed_at":"2026-07-06T18:34:24.078145Z","submitted_at":"2024-06-20T17:26:01Z","title":"MMBench-Video: A Long-Form Multi-Shot Benchmark for Holistic Video Understanding","version":3},"cited_work":{"arxiv_id":"2406.14515","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.14515","snapshot_observed_at":"2026-06-28T23:12:46.636950Z","title":"Mmbench-video: A long-form multi-shot benchmark for holistic video under- standing","venue":null,"work_id":"0108342f-f183-4a3e-930e-337bd83a6601","year":2024},"citing_paper":{"arxiv_id":"2412.05271","last_updated":"2025-09-26T12:52:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-06T18:57:08Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","version":5},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-10T13:23:57.588851Z"},"links":{"cited_paper":"/paper/2406.14515","citing_paper":"/paper/2412.05271"},"observation_digest":"sha256:b66c9cbbe22649157e86f38a835d60890648497dab6c031fcb961099b9c30fef","observation_id":"0bbc87ab-4cc4-4e82-851a-e02b4159ce21","resolution":{"observed_at":"2026-05-10T13:23:57.987120Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14515","last_updated":"2024-10-30T13:38:10Z","snapshot_observed_at":"2026-07-06T18:34:24.078145Z","submitted_at":"2024-06-20T17:26:01Z","title":"MMBench-Video: A Long-Form Multi-Shot Benchmark for Holistic Video Understanding","version":3},"cited_work":{"arxiv_id":"2406.14515","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.14515","snapshot_observed_at":"2026-06-28T23:12:46.636950Z","title":"Mmbench-video: A long-form multi-shot benchmark for holistic video under- standing","venue":null,"work_id":"0108342f-f183-4a3e-930e-337bd83a6601","year":2024},"citing_paper":{"arxiv_id":"2412.17574","last_updated":"2026-04-13T15:05:36Z","snapshot_observed_at":"2026-07-06T20:12:09.120832Z","submitted_at":"2024-12-23T13:45:56Z","title":"HumanVBench: Probing Human-Centric Video Understanding in MLLMs with Automatically Synthesized Benchmarks","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-23T07:05:08.716223Z"},"links":{"cited_paper":"/paper/2406.14515","citing_paper":"/paper/2412.17574"},"observation_digest":"sha256:c168b630404843633e3e8dcca6d75d4d04a72a9872c2f5dc9354475ed8c04048","observation_id":"44f54430-2512-4e8a-a767-9125548171aa","resolution":{"observed_at":"2026-05-23T07:05:29.189875Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14515","last_updated":"2024-10-30T13:38:10Z","snapshot_observed_at":"2026-07-06T18:34:24.078145Z","submitted_at":"2024-06-20T17:26:01Z","title":"MMBench-Video: A Long-Form Multi-Shot Benchmark for Holistic Video Understanding","version":3},"cited_work":{"arxiv_id":"2406.14515","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.14515","snapshot_observed_at":"2026-06-28T23:12:46.636950Z","title":"Mmbench-video: A long-form multi-shot benchmark for holistic video under- standing","venue":null,"work_id":"0108342f-f183-4a3e-930e-337bd83a6601","year":2024},"citing_paper":{"arxiv_id":"2501.04001","last_updated":"2025-11-03T17:35:29Z","snapshot_observed_at":"2026-08-08T01:58:42.644918Z","submitted_at":"2025-01-07T18:58:54Z","title":"Sa2VA: Marrying SAM2 with LLaVA for Dense Grounded Understanding of Images and Videos","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-16T11:39:22.340737Z"},"links":{"cited_paper":"/paper/2406.14515","citing_paper":"/paper/2501.04001"},"observation_digest":"sha256:f55c98a3d47a0d3a28174a10fc26df5b04cdf1b2761cfb62d7e264126ea7b62d","observation_id":"be82f068-0bf9-414b-8437-76fa6d796041","resolution":{"observed_at":"2026-05-16T11:39:22.438229Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14515","last_updated":"2024-10-30T13:38:10Z","snapshot_observed_at":"2026-07-06T18:34:24.078145Z","submitted_at":"2024-06-20T17:26:01Z","title":"MMBench-Video: A Long-Form Multi-Shot Benchmark for Holistic Video Understanding","version":3},"cited_work":{"arxiv_id":"2406.14515","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.14515","snapshot_observed_at":"2026-06-28T23:12:46.636950Z","title":"Mmbench-video: A long-form multi-shot benchmark for holistic video under- standing","venue":null,"work_id":"0108342f-f183-4a3e-930e-337bd83a6601","year":2024},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2406.14515","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:3899fb7176536064bca1d5a101e883448940e12a1991dddc5449fb399753b521","observation_id":"bcd3f53a-9953-4561-8386-c85bc699baec","resolution":{"observed_at":"2026-05-17T02:52:20.710199Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14515","last_updated":"2024-10-30T13:38:10Z","snapshot_observed_at":"2026-07-06T18:34:24.078145Z","submitted_at":"2024-06-20T17:26:01Z","title":"MMBench-Video: A Long-Form Multi-Shot Benchmark for Holistic Video Understanding","version":3},"cited_work":{"arxiv_id":"2406.14515","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.14515","snapshot_observed_at":"2026-06-28T23:12:46.636950Z","title":"Mmbench-video: A long-form multi-shot benchmark for holistic video under- standing","venue":null,"work_id":"0108342f-f183-4a3e-930e-337bd83a6601","year":2024},"citing_paper":{"arxiv_id":"2501.13826","last_updated":"2025-01-23T16:51:47Z","snapshot_observed_at":"2026-07-06T20:25:03.950783Z","submitted_at":"2025-01-23T16:51:47Z","title":"Video-MMMU: Evaluating Knowledge Acquisition from Multi-Discipline Professional Videos","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-14T00:32:41.059558Z"},"links":{"cited_paper":"/paper/2406.14515","citing_paper":"/paper/2501.13826"},"observation_digest":"sha256:dc2e38dc3077e5695bab81eff005d61a14911de9604aa90bf71c8b378779e561","observation_id":"ec1e6ef3-fe23-4b98-b54c-6247ef07148a","resolution":{"observed_at":"2026-05-14T00:32:41.214239Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14515","last_updated":"2024-10-30T13:38:10Z","snapshot_observed_at":"2026-07-06T18:34:24.078145Z","submitted_at":"2024-06-20T17:26:01Z","title":"MMBench-Video: A Long-Form Multi-Shot Benchmark for Holistic Video Understanding","version":3},"cited_work":{"arxiv_id":"2406.14515","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.14515","snapshot_observed_at":"2026-06-28T23:12:46.636950Z","title":"Mmbench-video: A long-form multi-shot benchmark for holistic video under- standing","venue":null,"work_id":"0108342f-f183-4a3e-930e-337bd83a6601","year":2024},"citing_paper":{"arxiv_id":"2502.04326","last_updated":"2026-03-01T04:35:41Z","snapshot_observed_at":"2026-07-06T20:32:23.502480Z","submitted_at":"2025-02-06T18:59:40Z","title":"WorldSense: Evaluating Real-world Omnimodal Understanding for Multimodal LLMs","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-17T05:53:26.066674Z"},"links":{"cited_paper":"/paper/2406.14515","citing_paper":"/paper/2502.04326"},"observation_digest":"sha256:d364937d1f561cd206b62c45d76aae719bd178e21eefb587011eebba562a55d0","observation_id":"fd4d1967-f873-4294-9409-954ecfc5c1bf","resolution":{"observed_at":"2026-05-17T05:53:26.306282Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14515","last_updated":"2024-10-30T13:38:10Z","snapshot_observed_at":"2026-07-06T18:34:24.078145Z","submitted_at":"2024-06-20T17:26:01Z","title":"MMBench-Video: A Long-Form Multi-Shot Benchmark for Holistic Video Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.14515","snapshot_observed_at":"2026-08-07T21:57:23.556854Z","title":"Mmbench-video: A long-form multi- shot benchmark for holistic video understanding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.09325","last_updated":"2025-02-13T13:38:17Z","snapshot_observed_at":"2026-08-07T21:50:11.513652Z","submitted_at":"2025-02-13T13:38:17Z","title":"A Benchmark for Crime Surveillance Video Analysis with Large Models","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T21:57:23.556854Z"},"links":{"cited_paper":"/paper/2406.14515","citing_paper":"/paper/2502.09325"},"observation_digest":"sha256:3b016f5bbf64eaf093bd9dfd10910343c6f4d1a07f1a792f1b7317d125f8c7dc","observation_id":"ccd52aed-5402-4615-9018-e7e59f80c64b","resolution":{"observed_at":"2026-08-07T21:57:23.556854Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14515","last_updated":"2024-10-30T13:38:10Z","snapshot_observed_at":"2026-07-06T18:34:24.078145Z","submitted_at":"2024-06-20T17:26:01Z","title":"MMBench-Video: A Long-Form Multi-Shot Benchmark for Holistic Video Understanding","version":3},"cited_work":{"arxiv_id":"2406.14515","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.14515","snapshot_observed_at":"2026-06-28T23:12:46.636950Z","title":"Mmbench-video: A long-form multi-shot benchmark for holistic video under- standing","venue":null,"work_id":"0108342f-f183-4a3e-930e-337bd83a6601","year":2024},"citing_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-23T02:25:04.405036Z"},"links":{"cited_paper":"/paper/2406.14515","citing_paper":"/paper/2502.13923"},"observation_digest":"sha256:ef873183b042e19614297fc93076724cecb9c4bd91a607f78627e90cc54a9995","observation_id":"3129f62a-b5cf-49d4-8aaa-2cf1e10f6c44","resolution":{"observed_at":"2026-05-23T02:25:19.026892Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14515","last_updated":"2024-10-30T13:38:10Z","snapshot_observed_at":"2026-07-06T18:34:24.078145Z","submitted_at":"2024-06-20T17:26:01Z","title":"MMBench-Video: A Long-Form Multi-Shot Benchmark for Holistic Video Understanding","version":3},"cited_work":{"arxiv_id":"2406.14515","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.14515","snapshot_observed_at":"2026-06-28T23:12:46.636950Z","title":"Mmbench-video: A long-form multi-shot benchmark for holistic video under- standing","venue":null,"work_id":"0108342f-f183-4a3e-930e-337bd83a6601","year":2024},"citing_paper":{"arxiv_id":"2504.10479","last_updated":"2025-04-19T03:47:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-14T17:59:25Z","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-10T13:41:07.991012Z"},"links":{"cited_paper":"/paper/2406.14515","citing_paper":"/paper/2504.10479"},"observation_digest":"sha256:dd94b272ca2606d382b208a5cb09263113cc8d57df52f8142abc44edc918f52b","observation_id":"a6b6f22f-c849-40be-aee7-4c23dfdc372a","resolution":{"observed_at":"2026-05-10T13:41:08.272679Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14515","last_updated":"2024-10-30T13:38:10Z","snapshot_observed_at":"2026-07-06T18:34:24.078145Z","submitted_at":"2024-06-20T17:26:01Z","title":"MMBench-Video: A Long-Form Multi-Shot Benchmark for Holistic Video Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.14515","snapshot_observed_at":"2026-08-07T12:45:44.920605Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23693","last_updated":"2025-05-29T17:31:13Z","snapshot_observed_at":"2026-08-07T18:41:28.769570Z","submitted_at":"2025-05-29T17:31:13Z","title":"VF-Eval: Evaluating Multimodal LLMs for Generating Feedback on AIGC Videos","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-07T12:45:44.920605Z"},"links":{"cited_paper":"/paper/2406.14515","citing_paper":"/paper/2505.23693"},"observation_digest":"sha256:e8d7f15627320518f9de2a08d4961becb7fecd28e57782520a035914e96bf3bf","observation_id":"2b773435-f45d-44d3-a2a5-28ef34363b23","resolution":{"observed_at":"2026-08-07T12:45:44.920605Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14515","last_updated":"2024-10-30T13:38:10Z","snapshot_observed_at":"2026-07-06T18:34:24.078145Z","submitted_at":"2024-06-20T17:26:01Z","title":"MMBench-Video: A Long-Form Multi-Shot Benchmark for Holistic Video Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.14515","snapshot_observed_at":"2026-08-07T06:00:56.863411Z","title":"Mmbench-video: A long-form multi-shot benchmark for holistic video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06275","last_updated":"2025-06-06T17:58:36Z","snapshot_observed_at":"2026-08-07T21:48:03.079145Z","submitted_at":"2025-06-06T17:58:36Z","title":"Movie Facts and Fibs (MF$^2$): A Benchmark for Long Movie Understanding","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T06:00:56.863411Z"},"links":{"cited_paper":"/paper/2406.14515","citing_paper":"/paper/2506.06275"},"observation_digest":"sha256:75fa6d1189fa998e7c204dbb230fd9449d643bc5a72c092102dd3b4923a6b2a0","observation_id":"e8a93478-79eb-43dc-9b5d-4b9eed8e11e0","resolution":{"observed_at":"2026-08-07T06:00:56.863411Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14515","last_updated":"2024-10-30T13:38:10Z","snapshot_observed_at":"2026-07-06T18:34:24.078145Z","submitted_at":"2024-06-20T17:26:01Z","title":"MMBench-Video: A Long-Form Multi-Shot Benchmark for Holistic Video Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.14515","snapshot_observed_at":"2026-08-06T21:59:15.950597Z","title":"arXiv preprint arXiv:2406.14515","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.02946","last_updated":"2025-06-28T15:24:05Z","snapshot_observed_at":"2026-08-08T00:43:49.554763Z","submitted_at":"2025-06-28T15:24:05Z","title":"Iterative Zoom-In: Temporal Interval Exploration for Long Video Understanding","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-06T21:59:15.950597Z"},"links":{"cited_paper":"/paper/2406.14515","citing_paper":"/paper/2507.02946"},"observation_digest":"sha256:8e4c00066d6668c2e337697a5f391fde71f00c6b2eaf2b5a18b2737de65727ea","observation_id":"07a3c23f-0848-4e87-9a93-13c1267286ba","resolution":{"observed_at":"2026-08-06T21:59:15.950597Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14515","last_updated":"2024-10-30T13:38:10Z","snapshot_observed_at":"2026-07-06T18:34:24.078145Z","submitted_at":"2024-06-20T17:26:01Z","title":"MMBench-Video: A Long-Form Multi-Shot Benchmark for Holistic Video Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.14515","snapshot_observed_at":"2026-08-06T18:03:13.983556Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-07T00:34:11.228009Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:13.983556Z"},"links":{"cited_paper":"/paper/2406.14515","citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:796473884583f0c8538e900dd484be5a94335aa568f53c9316d3b5c72ed71fea","observation_id":"2b04f4a6-1cdd-49c6-a73d-34ebafa03d10","resolution":{"observed_at":"2026-08-06T18:03:13.983556Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14515","last_updated":"2024-10-30T13:38:10Z","snapshot_observed_at":"2026-07-06T18:34:24.078145Z","submitted_at":"2024-06-20T17:26:01Z","title":"MMBench-Video: A Long-Form Multi-Shot Benchmark for Holistic Video Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.14515","snapshot_observed_at":"2026-08-06T15:48:31.569319Z","title":"Mmbench-video: A long-form multi-shot benchmark for holistic video under- standing","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:31.569319Z"},"links":{"cited_paper":"/paper/2406.14515","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:c71d9da35b652292e7590b6a2da23973ae8ce16279914293ca4adef9f7804d81","observation_id":"419a10bf-2fc6-4e6e-a497-af875e6cbe22","resolution":{"observed_at":"2026-08-06T15:48:31.569319Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14515","last_updated":"2024-10-30T13:38:10Z","snapshot_observed_at":"2026-07-06T18:34:24.078145Z","submitted_at":"2024-06-20T17:26:01Z","title":"MMBench-Video: A Long-Form Multi-Shot Benchmark for Holistic Video Understanding","version":3},"cited_work":{"arxiv_id":"2406.14515","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.14515","snapshot_observed_at":"2026-06-28T23:12:46.636950Z","title":"Mmbench-video: A long-form multi-shot benchmark for holistic video under- standing","venue":null,"work_id":"0108342f-f183-4a3e-930e-337bd83a6601","year":2024},"citing_paper":{"arxiv_id":"2508.18265","last_updated":"2025-08-27T14:39:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-25T17:58:17Z","title":"InternVL3.5: Advancing Open-Source Multimodal Models in Versatility, Reasoning, and Efficiency","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-10T11:58:58.660564Z"},"links":{"cited_paper":"/paper/2406.14515","citing_paper":"/paper/2508.18265"},"observation_digest":"sha256:2db7d035360a0ba7095cd09ec167150d340f94abf8cf7a505d430c17adc7681d","observation_id":"37ab95b4-f970-4028-bb65-c12a90eb0e61","resolution":{"observed_at":"2026-05-10T11:58:59.080794Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14515","last_updated":"2024-10-30T13:38:10Z","snapshot_observed_at":"2026-07-06T18:34:24.078145Z","submitted_at":"2024-06-20T17:26:01Z","title":"MMBench-Video: A Long-Form Multi-Shot Benchmark for Holistic Video Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.14515","snapshot_observed_at":"2026-08-03T02:37:20.820402Z","title":"arXiv preprint arXiv:2406.14515 (2024) 3","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.06828","last_updated":"2026-07-31T11:30:44Z","snapshot_observed_at":"2026-08-07T15:29:04.977312Z","submitted_at":"2026-03-06T19:43:26Z","title":"Step-Level Visual Grounding Faithfulness Predicts Out-of-Distribution Generalization in Long-Horizon Vision-Language Models","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-03T02:37:20.820402Z"},"links":{"cited_paper":"/paper/2406.14515","citing_paper":"/paper/2603.06828"},"observation_digest":"sha256:8657e50bfa7801e4cca42f95f6f52e8f4960af98d24c98b836d55134fbce1936","observation_id":"7943e501-1c3d-4caa-adb1-bb1002222058","resolution":{"observed_at":"2026-08-03T02:37:20.820402Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14515","last_updated":"2024-10-30T13:38:10Z","snapshot_observed_at":"2026-07-06T18:34:24.078145Z","submitted_at":"2024-06-20T17:26:01Z","title":"MMBench-Video: A Long-Form Multi-Shot Benchmark for Holistic Video Understanding","version":3},"cited_work":{"arxiv_id":"2406.14515","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.14515","snapshot_observed_at":"2026-06-28T23:12:46.636950Z","title":"Mmbench-video: A long-form multi-shot benchmark for holistic video under- standing","venue":null,"work_id":"0108342f-f183-4a3e-930e-337bd83a6601","year":2024},"citing_paper":{"arxiv_id":"2605.30673","last_updated":"2026-07-06T14:01:36Z","snapshot_observed_at":"2026-08-02T07:17:06.461529Z","submitted_at":"2026-05-29T00:06:54Z","title":"TeachObs: A Human-Validated Benchmark for Multimodal Teaching Observation and Model Evaluation","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-28T23:11:43.712391Z"},"links":{"cited_paper":"/paper/2406.14515","citing_paper":"/paper/2605.30673"},"observation_digest":"sha256:f2442eb7165087caa16eb91b8521c0e637fef4c4ca3cde6f870eb7c7caa7baf3","observation_id":"59fb7299-bb2e-4356-abf1-6aca0a378afb","resolution":{"observed_at":"2026-06-28T23:12:46.638751Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14515","last_updated":"2024-10-30T13:38:10Z","snapshot_observed_at":"2026-07-06T18:34:24.078145Z","submitted_at":"2024-06-20T17:26:01Z","title":"MMBench-Video: A Long-Form Multi-Shot Benchmark for Holistic Video Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.14515","snapshot_observed_at":"2026-08-01T22:38:46.251209Z","title":"arXiv preprint arXiv:2406.14515 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15689","last_updated":"2026-07-17T07:04:31Z","snapshot_observed_at":"2026-08-06T19:22:15.675454Z","submitted_at":"2026-07-17T07:04:31Z","title":"Efficient Frame Selection for Long Videos at Test Time with Attention-Based MLLM Selectors","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-01T22:38:46.251209Z"},"links":{"cited_paper":"/paper/2406.14515","citing_paper":"/paper/2607.15689"},"observation_digest":"sha256:1f2101a8e9d8098349c6e13c74b659336ef03bf4ab16b2aad4d7f8edc0151c03","observation_id":"08fc3bc3-eab0-45af-a13f-60e86cd4ff4e","resolution":{"observed_at":"2026-08-01T22:38:46.251209Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2406.14515/citation-record","integrity":"/paper/2406.14515/integrity","json":"/paper/2406.14515/citation-record.json","paper":"/paper/2406.14515"},"outbound":[],"paper":{"arxiv_id":"2406.14515","last_updated":"2024-10-30T13:38:10Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T18:34:24.078145Z","submitted_at":"2024-06-20T17:26:01Z","title":"MMBench-Video: A Long-Form Multi-Shot Benchmark for Holistic Video Understanding"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 20 inbound Pith citation observations for arXiv:2406.14515."}