{"as_of":"2026-08-11T13:20:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:dfd1a1141a1bc0ef47457e84b9b0a7397e7413857ab81f993499be53bb161a92","coverage":[{"denominator":104,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-17T20:22:34.954228Z","state":"measured"},{"denominator":151,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":151,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":51,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":51,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T11:48:07.183821Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-04T08:29:41.281374Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":"2311.17005","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-04T08:29:41.281374Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","venue":"cs.CV","work_id":"4ada183a-2d55-48d4-b8b7-a491b95e490e","year":2023},"citing_paper":{"arxiv_id":"2403.00476","last_updated":"2024-06-03T04:13:39Z","snapshot_observed_at":"2026-08-04T21:17:37.211833Z","submitted_at":"2024-03-01T12:02:19Z","title":"TempCompass: Do Video LLMs Really Understand Videos?","version":3},"reference_index":97,"source":"arxiv_source","source_observed_at":"2026-05-17T02:46:16.632743Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2403.00476"},"observation_digest":"sha256:e8f918190ef615f7c00c4a816257b75edd7ba3af5171a760e68a32af6add7371","observation_id":"2e4f28de-4321-4ddf-9b94-5d5d7ce4b5bf","resolution":{"observed_at":"2026-05-17T20:22:35.388231Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":"2311.17005","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-04T08:29:41.281374Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","venue":"cs.CV","work_id":"4ada183a-2d55-48d4-b8b7-a491b95e490e","year":2023},"citing_paper":{"arxiv_id":"2404.16994","last_updated":"2024-04-29T14:52:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-25T19:29:55Z","title":"PLLaVA : Parameter-free LLaVA Extension from Images to Videos for Video Dense Captioning","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-15T20:21:57.873354Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2404.16994"},"observation_digest":"sha256:968d8d9d97d174fd20bea755f086d91cffc57fb71e94e590a8710d61b43ba698","observation_id":"78825fca-ed50-493d-b9f5-23a9dfe39e04","resolution":{"observed_at":"2026-05-17T20:22:35.388231Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":"2311.17005","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-04T08:29:41.281374Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","venue":"cs.CV","work_id":"4ada183a-2d55-48d4-b8b7-a491b95e490e","year":2023},"citing_paper":{"arxiv_id":"2406.04264","last_updated":"2025-01-01T15:53:58Z","snapshot_observed_at":"2026-08-03T20:38:36.602554Z","submitted_at":"2024-06-06T17:09:32Z","title":"MLVU: Benchmarking Multi-task Long Video Understanding","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-14T19:55:26.333923Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2406.04264"},"observation_digest":"sha256:0ea68f40f7e9ae33ff47c6de1560379efa5a17c917222480e478bf43755eb391","observation_id":"24bf758a-fda9-4c5e-a094-156e14d055f9","resolution":{"observed_at":"2026-05-17T20:22:35.388231Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":"2311.17005","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-04T08:29:41.281374Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","venue":"cs.CV","work_id":"4ada183a-2d55-48d4-b8b7-a491b95e490e","year":2023},"citing_paper":{"arxiv_id":"2406.08035","last_updated":"2025-08-09T10:54:59Z","snapshot_observed_at":"2026-08-08T19:38:26.415599Z","submitted_at":"2024-06-12T09:36:52Z","title":"LVBench: An Extreme Long Video Understanding Benchmark","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-19T11:55:30.048525Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2406.08035"},"observation_digest":"sha256:6d8927fe3bc8ca71f3765a156077443ea5a4962d78cf1ac2655e0320c5dd0f23","observation_id":"54cf1153-ec5d-4c64-bf86-2e4f89809966","resolution":{"observed_at":"2026-05-19T11:55:30.185743Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":"2311.17005","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-04T08:29:41.281374Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","venue":"cs.CV","work_id":"4ada183a-2d55-48d4-b8b7-a491b95e490e","year":2023},"citing_paper":{"arxiv_id":"2407.03320","last_updated":"2024-07-03T17:59:21Z","snapshot_observed_at":"2026-08-04T22:09:42.241578Z","submitted_at":"2024-07-03T17:59:21Z","title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-05-17T10:46:28.447347Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2407.03320"},"observation_digest":"sha256:47a295bad4543bbd6fd85a58f5d11fa913765762f15e2f1a1c81fb8c316037d3","observation_id":"29263d61-c4f2-49bf-956f-2bab10c7f3fc","resolution":{"observed_at":"2026-05-17T20:22:35.388231Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":"2311.17005","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-04T08:29:41.281374Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","venue":"cs.CV","work_id":"4ada183a-2d55-48d4-b8b7-a491b95e490e","year":2023},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":223,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:d0ebd42749db81fed6620277ffe03f9c29b6e963b6035a8cddb3a8921449f38e","observation_id":"3fa8ee51-1ac0-4c2d-b14f-feace5327b12","resolution":{"observed_at":"2026-05-20T06:20:36.427467Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-08-11T11:48:07.183821Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.14965","last_updated":"2025-01-11T14:08:22Z","snapshot_observed_at":"2026-08-11T11:43:15.303209Z","submitted_at":"2024-12-19T15:44:04Z","title":"Movie2Story: A framework for understanding videos and telling stories in the form of novel text","version":2},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-11T11:48:07.183821Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2412.14965"},"observation_digest":"sha256:a0a6e5319afb953a038efac831eaf868c1e60cd4e10e4b7da06acc33a0efdeb8","observation_id":"e5eb1154-0f5e-4f3b-97a6-4c520c7c4bb5","resolution":{"observed_at":"2026-08-11T11:48:07.183821Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-08-11T05:41:47.557078Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.17295","last_updated":"2024-12-23T05:32:48Z","snapshot_observed_at":"2026-08-11T05:34:54.756252Z","submitted_at":"2024-12-23T05:32:48Z","title":"Friends-MMC: A Dataset for Multi-modal Multi-party Conversation Understanding","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-11T05:41:47.557078Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2412.17295"},"observation_digest":"sha256:efbcbb33d14ea5048bd113b6b6e23c09f58d3f22fdbee349b086c003acdfee85","observation_id":"05de5a87-1bcb-4955-bec0-463ccbc8413c","resolution":{"observed_at":"2026-08-11T05:41:47.557078Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-08-10T22:57:40.056441Z","title":"Mvbench: A comprehensive multi-modal video understand- ing benchmark","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.00584","last_updated":"2025-04-17T10:10:16Z","snapshot_observed_at":"2026-08-11T01:33:05.827241Z","submitted_at":"2024-12-31T18:17:05Z","title":"Online Video Understanding: OVBench and VideoChat-Online","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-10T22:57:40.056441Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2501.00584"},"observation_digest":"sha256:ace6e0279abe21d2816d3ddb45693ce3c98d41ba7c8b99dc461ab0d4929d4098","observation_id":"38f105f4-5c9a-4fc0-bb55-2fb2def1c153","resolution":{"observed_at":"2026-08-10T22:57:40.056441Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-08-10T22:32:55.405872Z","title":"Mvbench: A comprehensive multi- modal video understanding benchmark","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-08-10T22:25:36.091919Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-10T22:32:55.405872Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2501.01428"},"observation_digest":"sha256:05b767719185ff4727763da89d60b64ec3dc796dcce69adb84d6ef4883aa1c84","observation_id":"f405959f-b157-46c2-a7ad-577c02bc2f9a","resolution":{"observed_at":"2026-08-10T22:32:55.405872Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-08-10T22:19:53.377380Z","title":"Mvbench: A comprehensive multi-modal video understand- ing benchmark","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.02135","last_updated":"2025-01-03T23:03:24Z","snapshot_observed_at":"2026-08-10T23:06:48.073542Z","submitted_at":"2025-01-03T23:03:24Z","title":"AVTrustBench: Assessing and Enhancing Reliability and Robustness in Audio-Visual LLMs","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-10T22:19:53.377380Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2501.02135"},"observation_digest":"sha256:6829cad84fe33cc0032da8634ce54791f51586b809475e421c65981a5173f2e7","observation_id":"118b31ef-01dd-4a1f-88c9-4979b156b259","resolution":{"observed_at":"2026-08-10T22:19:53.377380Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-08-10T22:08:09.460300Z","title":"Mvbench: A comprehensive multi-modal video understanding benchmark,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.02765","last_updated":"2025-01-06T05:15:59Z","snapshot_observed_at":"2026-08-11T04:56:13.579583Z","submitted_at":"2025-01-06T05:15:59Z","title":"Visual Large Language Models for Generalized and Specialized Applications","version":1},"reference_index":147,"source":"pdf_text","source_observed_at":"2026-08-10T22:08:09.460300Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2501.02765"},"observation_digest":"sha256:cb88ecd242f0d77574f3a1526cb4ed741dcf7f25562510e7c204fa1701ee7d40","observation_id":"9545123b-93dc-455a-a7da-6026fec7ccc5","resolution":{"observed_at":"2026-08-10T22:08:09.460300Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":"2311.17005","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-04T08:29:41.281374Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","venue":"cs.CV","work_id":"4ada183a-2d55-48d4-b8b7-a491b95e490e","year":2023},"citing_paper":{"arxiv_id":"2501.13106","last_updated":"2025-06-03T03:33:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T18:59:46Z","title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","version":4},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-11T01:19:59.603343Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2501.13106"},"observation_digest":"sha256:58dba8be9aa07b69984796df09433c330c5ba37e12b25f192839c69836b1270f","observation_id":"c3c770c7-f02e-41f2-97e0-60799e2bb24a","resolution":{"observed_at":"2026-05-17T20:22:35.388231Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-08-09T21:21:45.664685Z","title":"Mvbench: A comprehensive multi-modal video understanding benchmark, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.19098","last_updated":"2025-05-19T10:18:07Z","snapshot_observed_at":"2026-08-10T11:57:26.754523Z","submitted_at":"2025-01-31T12:45:46Z","title":"$\\infty$-Video: A Training-Free Approach to Long Video Understanding via Continuous-Time Memory Consolidation","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-09T21:21:45.664685Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2501.19098"},"observation_digest":"sha256:0a04f310c6561b51ec0717fc363781a7ccbe876456e79dbf91af246698a037f0","observation_id":"2806aa15-9fd4-43c2-a6cc-a52db8a1468e","resolution":{"observed_at":"2026-08-09T21:21:45.664685Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":"2311.17005","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-04T08:29:41.281374Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","venue":"cs.CV","work_id":"4ada183a-2d55-48d4-b8b7-a491b95e490e","year":2023},"citing_paper":{"arxiv_id":"2504.01805","last_updated":"2025-05-21T09:38:44Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-02T15:12:17Z","title":"SpaceR: Reinforcing MLLMs in Video Spatial Reasoning","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-15T15:18:43.724432Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2504.01805"},"observation_digest":"sha256:b1cc0b9242d952730f4b84e18c4debed153acb5271e1c283fdc98d23e387c7cf","observation_id":"0c98bf56-fdcf-4791-9623-3cec2e25a9f9","resolution":{"observed_at":"2026-05-17T20:22:35.388231Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-08-07T10:28:51.806239Z","title":"Mvbench: A comprehensive multi-modal video understanding benchmark","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05260","last_updated":"2025-06-05T17:21:16Z","snapshot_observed_at":"2026-08-09T11:16:28.299508Z","submitted_at":"2025-06-05T17:21:16Z","title":"LeanPO: Lean Preference Optimization for Likelihood Alignment in Video-LLMs","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:51.806239Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2506.05260"},"observation_digest":"sha256:3e04b2b00fa4f6447fac90c798ecde726050baff9429f0d24d9e67b597f3658f","observation_id":"ded1134d-8f6e-40d4-bcd3-355b0f706f3f","resolution":{"observed_at":"2026-08-07T10:28:51.806239Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-08-06T18:10:12.662954Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09068","last_updated":"2025-07-23T13:06:44Z","snapshot_observed_at":"2026-08-11T10:43:34.020108Z","submitted_at":"2025-07-11T23:07:04Z","title":"Infinite Video Understanding","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T18:10:12.662954Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2507.09068"},"observation_digest":"sha256:52be382dac77aadc3cf31cba8c81ad8855e7f39c47215cc8903b5b71fe7b86e6","observation_id":"cc6d9d5e-c795-46dd-b38d-5aa40d4a9983","resolution":{"observed_at":"2026-08-06T18:10:12.662954Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-08-06T14:36:55.854748Z","title":"Mvbench: A comprehen- sive multi-modal video understanding benchmark,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.18447","last_updated":"2025-07-24T14:33:06Z","snapshot_observed_at":"2026-08-06T19:31:11.014579Z","submitted_at":"2025-07-24T14:33:06Z","title":"PDB-Eval: An Evaluation of Large Multimodal Models for Description and Explanation of Personalized Driving Behavior","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T14:36:55.854748Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2507.18447"},"observation_digest":"sha256:c55e83d9a0b7c2857106036b9b3e24f162927c3cfac80288d5f6b74b2bca948f","observation_id":"c4787365-ae0f-474c-8cbe-961b343e0487","resolution":{"observed_at":"2026-08-06T14:36:55.854748Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":"2311.17005","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-04T08:29:41.281374Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","venue":"cs.CV","work_id":"4ada183a-2d55-48d4-b8b7-a491b95e490e","year":2023},"citing_paper":{"arxiv_id":"2507.21420","last_updated":"2026-04-28T19:54:45Z","snapshot_observed_at":"2026-08-11T08:12:42.121511Z","submitted_at":"2025-07-29T01:07:09Z","title":"ReGATE: Learning Faster and Better with Fewer Tokens in MLLMs","version":3},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-05-19T03:18:11.993413Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2507.21420"},"observation_digest":"sha256:2c605d6f4a8bf80ccdac5350764f23f55d3db1bf393b7590e12a1bb5d9c8983d","observation_id":"0486619b-47e8-490b-ba3d-a4906dbf350b","resolution":{"observed_at":"2026-05-19T03:22:01.214250Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-08-05T10:34:18.342680Z","title":"Preprint, arXiv:2311.17005","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.03986","last_updated":"2025-09-04T08:13:06Z","snapshot_observed_at":"2026-08-08T09:08:20.770745Z","submitted_at":"2025-09-04T08:13:06Z","title":"Promptception: How Sensitive Are Large Multimodal Models to Prompts?","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-05T10:34:18.342680Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2509.03986"},"observation_digest":"sha256:00099a20617da8c0bb364a25d536d0032058285f3814139cacf51c58c752bd26","observation_id":"fe55c291-66a1-444b-b22a-99fd953b47ed","resolution":{"observed_at":"2026-08-05T10:34:18.342680Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":"2311.17005","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-04T08:29:41.281374Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","venue":"cs.CV","work_id":"4ada183a-2d55-48d4-b8b7-a491b95e490e","year":2023},"citing_paper":{"arxiv_id":"2601.14724","last_updated":"2026-05-07T12:10:26Z","snapshot_observed_at":"2026-08-05T03:02:47.238051Z","submitted_at":"2026-01-21T07:26:15Z","title":"HERMES: KV Cache as Hierarchical Memory for Efficient Streaming Video Understanding","version":4},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-16T12:55:04.564442Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2601.14724"},"observation_digest":"sha256:8239e7e0c5cbf3ecb08875ed5f03f3c31f1f5caf09df6e296a97d0fbf9649234","observation_id":"dec76cd9-6cc1-4bdc-b315-c7a48ee6782e","resolution":{"observed_at":"2026-05-17T20:22:35.388231Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-08-03T02:37:22.472754Z","title":"arXiv preprint arXiv:2311.17005 (2023) 3","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.06828","last_updated":"2026-07-31T11:30:44Z","snapshot_observed_at":"2026-08-07T15:29:04.977312Z","submitted_at":"2026-03-06T19:43:26Z","title":"Step-Level Visual Grounding Faithfulness Predicts Out-of-Distribution Generalization in Long-Horizon Vision-Language Models","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-03T02:37:22.472754Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2603.06828"},"observation_digest":"sha256:5e6a3bbb6b1d213e35e222ac935baf053f2848afc29699f5cbc1d3c429cbcbdb","observation_id":"30f66e08-de3e-44e7-a9d2-d0ad4ff761ee","resolution":{"observed_at":"2026-08-03T02:37:22.472754Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":"2311.17005","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-04T08:29:41.281374Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","venue":"cs.CV","work_id":"4ada183a-2d55-48d4-b8b7-a491b95e490e","year":2023},"citing_paper":{"arxiv_id":"2603.27259","last_updated":"2026-06-18T21:01:40Z","snapshot_observed_at":"2026-08-02T11:52:58.572026Z","submitted_at":"2026-03-28T12:44:19Z","title":"Seeing the Scene Matters: Revealing Forgetting in Video Understanding Models with a Scene-Aware Long-Video Benchmark","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-14T22:05:07.326202Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2603.27259"},"observation_digest":"sha256:634d19198c11a08551c5bc261589c87eb26f9d63a540e0895e5d55af4579951b","observation_id":"1b4f11f8-a14f-41fe-aba9-d21691f4ad69","resolution":{"observed_at":"2026-05-17T20:22:35.388231Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-13T15:29:31.834566Z","title":"arXiv preprint arXiv:2311.17005 (2024) 1, 2, 4","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.29943","last_updated":"2026-07-10T15:58:11Z","snapshot_observed_at":"2026-08-07T09:23:25.859554Z","submitted_at":"2026-03-31T16:16:17Z","title":"Diagnosing Long-Video Quantitative Reasoning in Multimodal LLMs via Enumeration and Counting","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-07-13T15:29:31.834566Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2603.29943"},"observation_digest":"sha256:f0ddef466ff0b65569ecd5bfef89639f30019036708310f4824300be31ded6aa","observation_id":"f3b5ea87-3001-41a2-b6e4-a5b1c3405525","resolution":{"observed_at":"2026-07-13T15:29:31.834566Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":"2311.17005","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-04T08:29:41.281374Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","venue":"cs.CV","work_id":"4ada183a-2d55-48d4-b8b7-a491b95e490e","year":2023},"citing_paper":{"arxiv_id":"2604.02467","last_updated":"2026-04-27T18:17:15Z","snapshot_observed_at":"2026-08-02T13:43:39.563153Z","submitted_at":"2026-04-02T18:58:56Z","title":"VERTIGO: Visual Preference Optimization for Cinematic Camera Trajectory Generation","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-13T21:14:42.021240Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2604.02467"},"observation_digest":"sha256:51517ef122852036e0c42421a47e66d15a94461e48e02b7099ae2fa544f3a149","observation_id":"5b4fec27-de6a-47d4-9f6a-aaa57272fab5","resolution":{"observed_at":"2026-05-17T20:22:35.388231Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":"2311.17005","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-04T08:29:41.281374Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","venue":"cs.CV","work_id":"4ada183a-2d55-48d4-b8b7-a491b95e490e","year":2023},"citing_paper":{"arxiv_id":"2604.08077","last_updated":"2026-04-09T10:48:32Z","snapshot_observed_at":"2026-08-11T04:13:42.078728Z","submitted_at":"2026-04-09T10:48:32Z","title":"AdaSpark: Adaptive Sparsity for Efficient Long-Video Understanding","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-10T17:21:47.439019Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2604.08077"},"observation_digest":"sha256:3712252bc5982bc2c5cda92483bfc3b5c92125574223cae3f2adda66a78ff6eb","observation_id":"15d9a4bb-9f80-4457-a5fc-f25e92342bdf","resolution":{"observed_at":"2026-05-17T20:22:35.388231Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":"2311.17005","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-04T08:29:41.281374Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","venue":"cs.CV","work_id":"4ada183a-2d55-48d4-b8b7-a491b95e490e","year":2023},"citing_paper":{"arxiv_id":"2604.08703","last_updated":"2026-04-09T18:51:16Z","snapshot_observed_at":"2026-07-06T22:57:45.084987Z","submitted_at":"2026-04-09T18:51:16Z","title":"QoS-QoE Translation with Large Language Model","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-10T17:01:57.703684Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2604.08703"},"observation_digest":"sha256:022674e082f6491549b660be9b7285d36c5ee186066ff36fc1851de244ab27c3","observation_id":"c03de85c-b9aa-4593-9672-80643601eb04","resolution":{"observed_at":"2026-05-17T20:22:35.388231Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":"2311.17005","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-04T08:29:41.281374Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","venue":"cs.CV","work_id":"4ada183a-2d55-48d4-b8b7-a491b95e490e","year":2023},"citing_paper":{"arxiv_id":"2604.16893","last_updated":"2026-04-18T07:56:32Z","snapshot_observed_at":"2026-07-06T23:04:05.813982Z","submitted_at":"2026-04-18T07:56:32Z","title":"EasyVideoR1: Easier RL for Video Understanding","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-10T07:41:27.231098Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2604.16893"},"observation_digest":"sha256:963447feb527414162b2c924ca94ecf67cc768e95ec95dbed9d15c89eebec422","observation_id":"ea0e2140-881b-4e1c-aa8b-ffc97f098bfc","resolution":{"observed_at":"2026-05-17T20:22:35.388231Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":"2311.17005","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-04T08:29:41.281374Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","venue":"cs.CV","work_id":"4ada183a-2d55-48d4-b8b7-a491b95e490e","year":2023},"citing_paper":{"arxiv_id":"2604.25185","last_updated":"2026-06-18T03:00:25Z","snapshot_observed_at":"2026-08-05T20:03:52.003085Z","submitted_at":"2026-04-28T03:43:53Z","title":"The category of Whittaker modules over the Cartan Type Lie algebra $\\bar{S}_2$","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-01T09:08:06.592577Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2604.25185"},"observation_digest":"sha256:cff0cd980b59b3199ba22cf2eefb8cdf24607269779c434b969c2f0312c50803","observation_id":"f3286376-a16a-49f5-a67c-e20e624b13ea","resolution":{"observed_at":"2026-07-01T09:25:40.462773Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":"2311.17005","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-04T08:29:41.281374Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","venue":"cs.CV","work_id":"4ada183a-2d55-48d4-b8b7-a491b95e490e","year":2023},"citing_paper":{"arxiv_id":"2604.25186","last_updated":"2026-04-30T03:30:43Z","snapshot_observed_at":"2026-08-10T21:39:03.625679Z","submitted_at":"2026-04-28T03:45:09Z","title":"FCMBench-Video: Benchmarking Document Video Intelligence","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-07T17:14:19.186123Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2604.25186"},"observation_digest":"sha256:b820cf2ea8bd5ebe0af2dba4fc28a0311efc84dfbf184bdf5024ce4db58ad3cc","observation_id":"6be03a38-c61e-4b96-81c1-35ca1a0d723f","resolution":{"observed_at":"2026-05-17T20:22:35.388231Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":"2311.17005","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-04T08:29:41.281374Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","venue":"cs.CV","work_id":"4ada183a-2d55-48d4-b8b7-a491b95e490e","year":2023},"citing_paper":{"arxiv_id":"2604.27083","last_updated":"2026-04-29T18:24:11Z","snapshot_observed_at":"2026-07-06T23:12:38.453388Z","submitted_at":"2026-04-29T18:24:11Z","title":"Co-Evolving Policy Distillation","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-07T08:23:41.819485Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2604.27083"},"observation_digest":"sha256:77c22cf162e8995d50e487887888e3c35af83067245179960da0562494c7614b","observation_id":"641ea536-c5c4-46f2-b3ad-6ea54775e99c","resolution":{"observed_at":"2026-05-17T20:22:35.388231Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":"2311.17005","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-04T08:29:41.281374Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","venue":"cs.CV","work_id":"4ada183a-2d55-48d4-b8b7-a491b95e490e","year":2023},"citing_paper":{"arxiv_id":"2605.03351","last_updated":"2026-05-05T04:13:32Z","snapshot_observed_at":"2026-07-06T23:16:15.509235Z","submitted_at":"2026-05-05T04:13:32Z","title":"VLMaxxing through FrameMogging Training-Free Anti-Recomputation for Video Vision-Language Models","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-08T01:30:15.463051Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2605.03351"},"observation_digest":"sha256:a7567a0d0fa575408c64d06ab00467b2bf366a35603a54b5b0768a5709a98116","observation_id":"bc72911b-dfb6-453a-91d8-f0bdd709f19f","resolution":{"observed_at":"2026-05-17T20:22:35.388231Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":"2311.17005","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-04T08:29:41.281374Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","venue":"cs.CV","work_id":"4ada183a-2d55-48d4-b8b7-a491b95e490e","year":2023},"citing_paper":{"arxiv_id":"2605.05225","last_updated":"2026-06-05T13:47:55Z","snapshot_observed_at":"2026-08-02T06:50:35.896043Z","submitted_at":"2026-04-19T07:25:39Z","title":"MACS: Modality-Aware Capacity Scaling for Efficient Multimodal MoE Inference","version":2},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-05-11T01:29:26.298131Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2605.05225"},"observation_digest":"sha256:679985a936dc0dc7325219bda81c28cbe25122b4b06233918be3c374d756155a","observation_id":"5809c465-a775-45a4-afe9-99d7cd7493e7","resolution":{"observed_at":"2026-05-17T20:22:35.388231Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":"2311.17005","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-04T08:29:41.281374Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","venue":"cs.CV","work_id":"4ada183a-2d55-48d4-b8b7-a491b95e490e","year":2023},"citing_paper":{"arxiv_id":"2605.19559","last_updated":"2026-05-19T09:02:20Z","snapshot_observed_at":"2026-08-10T16:22:29.533323Z","submitted_at":"2026-05-19T09:02:20Z","title":"EgoCoT-Bench: Benchmarking Grounded and Verifiable Operation-Centric Chain of Thought Reasoning for MLLMs","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-20T05:53:05.450946Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2605.19559"},"observation_digest":"sha256:e3c10664d5e79ff2a8a43ca5fe33eea83589676ad3b282dd345f682379252afd","observation_id":"eedde43e-b378-467f-8e9d-1b4d0196a6e8","resolution":{"observed_at":"2026-05-20T05:53:22.278869Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":"2311.17005","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-04T08:29:41.281374Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","venue":"cs.CV","work_id":"4ada183a-2d55-48d4-b8b7-a491b95e490e","year":2023},"citing_paper":{"arxiv_id":"2605.19950","last_updated":"2026-05-19T15:05:00Z","snapshot_observed_at":"2026-07-06T23:30:35.042915Z","submitted_at":"2026-05-19T15:05:00Z","title":"AffectVerse: Emotional World Models for Multimodal Affective Computing","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-20T06:46:33.612905Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2605.19950"},"observation_digest":"sha256:2378d1e2c87bfb3724edcebb88ccac2035f2e24650d9c8265446e664db215e0b","observation_id":"a260fcbd-c642-4343-975e-eeb7418c5157","resolution":{"observed_at":"2026-05-20T06:48:05.786338Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":"2311.17005","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-04T08:29:41.281374Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","venue":"cs.CV","work_id":"4ada183a-2d55-48d4-b8b7-a491b95e490e","year":2023},"citing_paper":{"arxiv_id":"2605.21625","last_updated":"2026-05-20T18:36:57Z","snapshot_observed_at":"2026-07-06T23:32:01.202542Z","submitted_at":"2026-05-20T18:36:57Z","title":"Flat-Pack Bench: Evaluating Spatio-Temporal Understanding in Large Vision-Language Models through Furniture Assembly","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-22T09:20:32.920925Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2605.21625"},"observation_digest":"sha256:5fd76440d6ade8c62ef4e0794cb8434a97cb85c8b9b45ec6f0c08e17df3e8f68","observation_id":"a521a322-6f9e-4180-8998-a1645aca73f5","resolution":{"observed_at":"2026-05-22T09:21:20.626071Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":"2311.17005","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-04T08:29:41.281374Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","venue":"cs.CV","work_id":"4ada183a-2d55-48d4-b8b7-a491b95e490e","year":2023},"citing_paper":{"arxiv_id":"2605.30673","last_updated":"2026-07-06T14:01:36Z","snapshot_observed_at":"2026-08-02T07:17:06.461529Z","submitted_at":"2026-05-29T00:06:54Z","title":"TeachObs: A Human-Validated Benchmark for Multimodal Teaching Observation and Model Evaluation","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-28T23:11:43.712391Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2605.30673"},"observation_digest":"sha256:fbb57c5ac40d5349d7a6670d26812344075cf558a119cd7dc1a82f28f16c5a46","observation_id":"a2f87306-3810-4368-a92c-4b59cea0e442","resolution":{"observed_at":"2026-06-28T23:12:46.646665Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":"2311.17005","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-04T08:29:41.281374Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","venue":"cs.CV","work_id":"4ada183a-2d55-48d4-b8b7-a491b95e490e","year":2023},"citing_paper":{"arxiv_id":"2606.03087","last_updated":"2026-06-02T03:17:34Z","snapshot_observed_at":"2026-07-06T23:43:22.513369Z","submitted_at":"2026-06-02T03:17:34Z","title":"Learning to Solve, Forgetting to Retain: Correct-Set Turnover in RLVR","version":1},"reference_index":113,"source":"arxiv_source","source_observed_at":"2026-06-28T11:50:00.954670Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2606.03087"},"observation_digest":"sha256:0af7831f3e9bb5b339269fc04cea0d213437794a33f31f6dc0253e5581b409b8","observation_id":"1e2857d9-8d63-4ab1-b226-c19bca24d5db","resolution":{"observed_at":"2026-07-02T01:36:25.117632Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":"2311.17005","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-04T08:29:41.281374Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","venue":"cs.CV","work_id":"4ada183a-2d55-48d4-b8b7-a491b95e490e","year":2023},"citing_paper":{"arxiv_id":"2606.05736","last_updated":"2026-06-04T05:55:15Z","snapshot_observed_at":"2026-07-06T23:45:42.379051Z","submitted_at":"2026-06-04T05:55:15Z","title":"VTI-CoT: Visual-Textual Interleaved Chain of Thought for Video Reasoning","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-28T01:52:44.785582Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2606.05736"},"observation_digest":"sha256:f3d5bb2b4a3da30cd31f15577913c617fe34b143da3426cefd7d718218bb03d3","observation_id":"3716f920-fa42-4ad1-966c-a3105a42bbe2","resolution":{"observed_at":"2026-07-02T12:46:56.826694Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":"2311.17005","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-04T08:29:41.281374Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","venue":"cs.CV","work_id":"4ada183a-2d55-48d4-b8b7-a491b95e490e","year":2023},"citing_paper":{"arxiv_id":"2606.18216","last_updated":"2026-06-16T17:46:02Z","snapshot_observed_at":"2026-08-10T09:12:54.601851Z","submitted_at":"2026-06-16T17:46:02Z","title":"Zone of Proximal Policy Optimization: Teacher in Prompts, Not Gradients","version":1},"reference_index":152,"source":"pdf_text","source_observed_at":"2026-06-27T01:08:52.981296Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2606.18216"},"observation_digest":"sha256:9d644cc24f865e921b2fcfd3ffcebe8323fc33576edb4e5acb15c45ba1d2ace0","observation_id":"fdcd9de7-c746-48ba-b93d-fb1a1da61a28","resolution":{"observed_at":"2026-07-03T20:48:55.970891Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":"2311.17005","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-04T08:29:41.281374Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","venue":"cs.CV","work_id":"4ada183a-2d55-48d4-b8b7-a491b95e490e","year":2023},"citing_paper":{"arxiv_id":"2606.19706","last_updated":"2026-06-18T02:05:14Z","snapshot_observed_at":"2026-07-06T23:54:56.685241Z","submitted_at":"2026-06-18T02:05:14Z","title":"NEST: Narrative Event Structures in Time for Long Video Understanding","version":1},"reference_index":281,"source":"arxiv_source","source_observed_at":"2026-06-26T17:57:55.366051Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2606.19706"},"observation_digest":"sha256:e02a255856863876d7b38eeee31c2056fb9ec6a28a8dbab693aaf25d92a2f2bd","observation_id":"ac9d4f6c-8f74-4f01-9182-ba005ea3cb3a","resolution":{"observed_at":"2026-07-04T03:29:31.178505Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":"2311.17005","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-04T08:29:41.281374Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","venue":"cs.CV","work_id":"4ada183a-2d55-48d4-b8b7-a491b95e490e","year":2023},"citing_paper":{"arxiv_id":"2606.21734","last_updated":"2026-06-19T20:43:49Z","snapshot_observed_at":"2026-08-05T18:05:51.515234Z","submitted_at":"2026-06-19T20:43:49Z","title":"HPP: Hierarchical Programmatic Probing for Long Video Understanding by Decoupling Perception and Reasoning","version":1},"reference_index":153,"source":"arxiv_source","source_observed_at":"2026-06-26T14:19:53.450263Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2606.21734"},"observation_digest":"sha256:bbc25dd9186f43c50db2c7fa757d032857371b4e93a2128bfc8a04260ca50351","observation_id":"61179361-398d-4af4-9299-cb6a3ca4c218","resolution":{"observed_at":"2026-07-04T06:39:37.664149Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":"2311.17005","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-04T08:29:41.281374Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","venue":"cs.CV","work_id":"4ada183a-2d55-48d4-b8b7-a491b95e490e","year":2023},"citing_paper":{"arxiv_id":"2606.22043","last_updated":"2026-06-20T13:48:19Z","snapshot_observed_at":"2026-08-10T08:54:14.180931Z","submitted_at":"2026-06-20T13:48:19Z","title":"When Does a Video-Language Model Stop Watching? Reward Strength Controls the Formation and Reversal of Visual Shortcuts in Multimodal RLVR","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-26T11:43:17.464276Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2606.22043"},"observation_digest":"sha256:786b69c430c5659e495a54adf10735c500bd73ec40a6791dac7d94edb6a7a36a","observation_id":"5b512191-e716-4f60-a9e1-5304ad24fb17","resolution":{"observed_at":"2026-07-04T08:29:41.282601Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":"2311.17005","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-04T08:29:41.281374Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","venue":"cs.CV","work_id":"4ada183a-2d55-48d4-b8b7-a491b95e490e","year":2023},"citing_paper":{"arxiv_id":"2606.24253","last_updated":"2026-06-26T09:37:54Z","snapshot_observed_at":"2026-08-11T09:58:16.500600Z","submitted_at":"2026-06-23T07:42:22Z","title":"TuringViT: Making SOTA Vision Transformers Accessible to All","version":2},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-06-29T05:32:26.746776Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2606.24253"},"observation_digest":"sha256:59f543fec939b3a073f2a4d345f97391469b978627c24c6a714ad9f2d03f3087","observation_id":"a3398f72-12fc-4496-a0a0-183f4d26b4dc","resolution":{"observed_at":"2026-06-29T15:03:32.148979Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":"2311.17005","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-04T08:29:41.281374Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","venue":"cs.CV","work_id":"4ada183a-2d55-48d4-b8b7-a491b95e490e","year":2023},"citing_paper":{"arxiv_id":"2606.29445","last_updated":"2026-06-28T15:11:19Z","snapshot_observed_at":"2026-08-02T18:05:44.581086Z","submitted_at":"2026-06-28T15:11:19Z","title":"Bridging VideoQA and Video-Guided Agentic Tasks via Generalized Keyframe Extraction","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-06-30T07:48:01.719339Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2606.29445"},"observation_digest":"sha256:542f2ec53b8106fc0ccfa52647eb7186dc54c670b985d30226f54bf9c960e3c9","observation_id":"995be05f-5f99-49d1-801c-27032a7c919d","resolution":{"observed_at":"2026-06-30T07:54:22.384015Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-07-11T08:59:46.244502Z","title":"arXiv preprint arXiv:2311.17005 (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.05089","last_updated":"2026-07-06T13:50:15Z","snapshot_observed_at":"2026-07-11T08:59:45.751461Z","submitted_at":"2026-07-06T13:50:15Z","title":"TimeThink: Reasoning with Time for Video LLMs","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-07-11T08:59:46.244502Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2607.05089"},"observation_digest":"sha256:2d2302f2e99ea37fadb804fb5f5452972ea3f583f400c569d9d813a5253ce20b","observation_id":"51f9cb7e-7a5c-4fb8-9a69-5c661e7ff162","resolution":{"observed_at":"2026-07-11T08:59:46.244502Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-08-02T02:42:46.866557Z","title":"Mvbench: A comprehensive multi- modal video understanding benchmark, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.14252","last_updated":"2026-07-15T18:12:28Z","snapshot_observed_at":"2026-08-07T21:41:10.578273Z","submitted_at":"2026-07-15T18:12:28Z","title":"MEMORA: Embodied Action Memory from Egocentric Videos for Reasoning and Planning","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-02T02:42:46.866557Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2607.14252"},"observation_digest":"sha256:0611710893c90046e28f5d37676efa343d2807d7364013ee5abd5147e7335be7","observation_id":"e3f76df0-2a30-4fb4-9156-65494a8acfce","resolution":{"observed_at":"2026-08-02T02:42:46.866557Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-08-04T23:46:51.590927Z","title":"IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR) , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.01644","last_updated":"2026-08-03T03:27:46Z","snapshot_observed_at":"2026-08-10T14:57:44.819541Z","submitted_at":"2026-08-03T03:27:46Z","title":"CRAFT: Compression via Recursive Adaptive Fusion of Video Tokens for Vision-Language Models","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-04T23:46:51.590927Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2608.01644"},"observation_digest":"sha256:e312dbdddea3924b51b2fd1eb6b6eacf3862c699b78216b21c221e432893e182","observation_id":"ff6fa96a-a337-43e2-b5ac-823cdd111a8b","resolution":{"observed_at":"2026-08-04T23:46:51.590927Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-08-08T20:16:34.442248Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.04302","last_updated":"2026-08-05T00:20:23Z","snapshot_observed_at":"2026-08-10T18:05:29.552100Z","submitted_at":"2026-08-05T00:20:23Z","title":"CLIP-CC-Bench: Evaluating Paragraph-Level Video Descriptions in Video-Language Models","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-08T20:16:34.442248Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2608.04302"},"observation_digest":"sha256:566f7c9fdf9fd4546ccbe692fbb6c8c044c49667cde82e682cc234fd1592ead8","observation_id":"a617dda4-f763-4c75-84bd-f36af7609cbb","resolution":{"observed_at":"2026-08-08T20:16:34.442248Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-08-07T04:24:54.025621Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.06361","last_updated":"2026-08-06T17:57:06Z","snapshot_observed_at":"2026-08-10T12:08:17.898559Z","submitted_at":"2026-08-06T17:57:06Z","title":"The Low Frequency Trap: Video Language Models Fail at Simple Event Bookkeeping","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-07T04:24:54.025621Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2608.06361"},"observation_digest":"sha256:7233e3256903980f814fc329aaf3fe0552e9d6c48cbfe1e82ccb5ceab22757e0","observation_id":"d91c9d61-57eb-42aa-9c96-71709166d49e","resolution":{"observed_at":"2026-08-07T04:24:54.025621Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.17005","snapshot_observed_at":"2026-08-10T21:18:34.635521Z","title":"Mvbench: A comprehensive multi-modal video under- standing benchmark.arXiv preprint arXiv:2311.17005, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.06756","last_updated":"2026-08-07T03:24:25Z","snapshot_observed_at":"2026-08-11T13:11:18.978309Z","submitted_at":"2026-08-07T03:24:25Z","title":"Capek 0.5: An Execution-Centric Vision-Language Model for Embodied Intelligence","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-10T21:18:34.635521Z"},"links":{"cited_paper":"/paper/2311.17005","citing_paper":"/paper/2608.06756"},"observation_digest":"sha256:57720569db5058df21a89e3cd7eaf6ea7d1b0346eb50d189a4cf3506e1d076e5","observation_id":"bd18a110-790c-4eee-bddb-ff6638e71253","resolution":{"observed_at":"2026-08-10T21:18:34.635521Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2311.17005/citation-record","integrity":"/paper/2311.17005/integrity","json":"/paper/2311.17005/citation-record.json","paper":"/paper/2311.17005"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2204.14198","last_updated":"2022-11-15T23:07:37Z","snapshot_observed_at":"2026-07-06T13:05:12.350238Z","submitted_at":"2022-04-29T16:29:01Z","title":"Flamingo: a Visual Language Model for Few-Shot Learning","version":2},"cited_work":{"arxiv_id":"2204.14198","doi":"10.48550/arxiv.2204.14198","metadata_source":"pith","pith_arxiv_id":"2204.14198","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Flamingo: a Visual Language Model for Few-Shot Learning","venue":"cs.CV","work_id":"a110f764-38dc-41b2-a802-53744ecea1fc","year":2022},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"cited_paper":"/paper/2204.14198","citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:ad0f999803bff94d38307d75fa58e8789b8b927c70054ee92a51b67c2239f4ab","observation_id":"47b0421e-9227-4d52-b309-3a812e2031e0","resolution":{"observed_at":"2026-05-17T20:22:35.111518Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12966","last_updated":"2023-10-13T02:41:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-24T17:59:17Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","version":3},"cited_work":{"arxiv_id":"2308.12966","doi":"10.48550/arxiv.2308.12966","metadata_source":"pith","pith_arxiv_id":"2308.12966","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","venue":"cs.CV","work_id":"cbc2bb21-b6bb-46c0-80bf-107e195ffe10","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"cited_paper":"/paper/2308.12966","citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:e26c91e4b9793dea287bc5329ffd450bc247eaca00d2d639b81a8eee8252372a","observation_id":"90f2ffa5-3409-475a-a4eb-0056df666e3f","resolution":{"observed_at":"2026-05-17T20:22:35.082058Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Frozen in time: A joint video and image encoder for end-to-end retrieval","venue":null,"work_id":"12562377-293a-4224-b83e-3f411bc1cd94","year":2021},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:4f70b01156fb6360599e31fadb52ed009423477eb8d54ab8ed5bc87e5ca19672","observation_id":"a781def4-bb44-416f-9b1c-4d89c54d6663","resolution":{"observed_at":"2026-05-17T20:22:35.287880Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"1b934981-5385-4e63-b823-9601505710bd","year":2019},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:eef98901fc4eaf942a8b5d5329961b3fa3ae7865e68c1089db38284d341a8e57","observation_id":"995854ba-d09c-4e4e-9e25-c06869744d6a","resolution":{"observed_at":"2026-05-17T20:22:35.290504Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Language models are few-shot learners","venue":null,"work_id":"82a86dd0-f3b8-4511-97b1-c7b263281e6e","year":2020},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:f42e28bb1a639498b59a5d95f94cc0a9340445a7191c84a609ab040a9f71c016","observation_id":"52cfa3b3-a378-41d7-8322-8ffb0a02d404","resolution":{"observed_at":"2026-05-17T20:22:35.293232Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Conceptual 12m: Pushing web-scale image-text pre-training to recognize long-tail visual concepts","venue":null,"work_id":"b37b3253-de04-4912-9bd7-64bbdbe12a2c","year":2021},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:0915127e69169ea07669978125cf4072dacb2bfe78c8faec769e2d37bba0b96c","observation_id":"c3e1ca57-0421-4234-88e3-88fda803c644","resolution":{"observed_at":"2026-05-17T20:22:35.296457Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Chen and William B","venue":null,"work_id":"3962f3f5-6017-4754-93cb-5bdbbea9fdbc","year":2011},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:16a478b06dc04d157b66ed2bd41bbbca79fcfa6018de4f731427c9697b9af05e","observation_id":"415a9afd-4921-4a85-96bb-8488c299c38e","resolution":{"observed_at":"2026-05-17T20:22:35.299565Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.09478","last_updated":"2023-11-07T18:25:48Z","snapshot_observed_at":"2026-08-06T12:38:03.720232Z","submitted_at":"2023-10-14T03:22:07Z","title":"MiniGPT-v2: large language model as a unified interface for vision-language multi-task learning","version":3},"cited_work":{"arxiv_id":"2310.09478","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.09478","snapshot_observed_at":"2026-07-08T22:35:40.586981Z","title":"MiniGPT-v2: large language model as a unified interface for vision-language multi-task learning","venue":"cs.CV","work_id":"fb62cd1b-3991-40be-a987-3cfa5772b5b5","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"cited_paper":"/paper/2310.09478","citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:34989438cd969e84b01b08b0c7e8a25cfdf99a69bec3b88ae30d35a460af980d","observation_id":"bf002db6-6836-4ce4-97d2-0bdbc78a7c41","resolution":{"observed_at":"2026-05-17T20:22:35.086538Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.15195","last_updated":"2023-07-03T16:08:00Z","snapshot_observed_at":"2026-07-06T15:47:07.545213Z","submitted_at":"2023-06-27T04:31:52Z","title":"Shikra: Unleashing Multimodal LLM's Referential Dialogue Magic","version":2},"cited_work":{"arxiv_id":"2306.15195","doi":"10.48550/arxiv.2306.15195","metadata_source":"pith","pith_arxiv_id":"2306.15195","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Shikra: Unleashing Multimodal LLM's Referential Dialogue Magic","venue":"cs.CV","work_id":"44525076-312a-4259-b79c-134cd7eeb297","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"cited_paper":"/paper/2306.15195","citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:67d56fff17cea03d0a26f3303f991af2729244ff5732707d5f61b171732147dc","observation_id":"5b8838c0-e6d1-4221-b067-239d76fb8ecd","resolution":{"observed_at":"2026-05-17T20:22:35.041933Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Shazeer, Vinodkumar Prab- hakaran, Emily Reif, Nan Du, Benton C","venue":null,"work_id":"e4033610-24b4-48e6-a9a8-9e0098e46be0","year":null},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:96f6b2e1cd7fde5250e174135b79729292e32ac6b20721f7cf7ef681e316da26","observation_id":"0613568e-bef9-4bb2-9dde-031db678094c","resolution":{"observed_at":"2026-05-17T20:22:35.302838Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"e2cd4575-4a6f-4b78-ae58-09ef41b7e7bc","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:3b944a292f03476e180e88e0e30409f317ebb3cfde2c5da1314241a2ddedb142","observation_id":"b185b69e-4a16-4aa6-a4e2-e8a43e68198a","resolution":{"observed_at":"2026-05-17T20:22:35.305479Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Fu, Stefano Ermon, Atri Rudra, and Christopher R´e","venue":null,"work_id":"ac438dfa-4634-48e3-a146-ccd102cf68d2","year":2022},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:0e7016c248e63029653d6b7504d875caf8635be932af7f70d06e38ef3a607598","observation_id":"4359a529-2860-4cb3-9520-561c8dbb8953","resolution":{"observed_at":"2026-05-17T20:22:35.308195Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Doell, and Jason J","venue":null,"work_id":"ea0b47a9-3ed6-49ce-94d6-bedb2fc55987","year":2013},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:2711693470047a0d4966928e1840259765c8671ec4a2b37637bf47092f6aab29","observation_id":"c48d2239-c31a-40c2-971a-a003b75eba68","resolution":{"observed_at":"2026-05-17T20:22:35.310868Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Imagenet: A large-scale hierarchical im- age database","venue":null,"work_id":"f4414a52-6972-452a-9c54-ff68d71d0fe2","year":2009},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:3409166cc4ff656d9fdf8df21becff4261b3c812145f12d4e7d75be3da283ab0","observation_id":"0e34e3db-000d-468f-8351-88cef33497d4","resolution":{"observed_at":"2026-05-17T20:22:35.313831Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1810.04805","last_updated":"2019-05-24T20:37:26Z","snapshot_observed_at":"2026-07-30T09:12:38.100527Z","submitted_at":"2018-10-11T00:50:01Z","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","version":2},"cited_work":{"arxiv_id":"1810.04805","doi":"10.1111/jofi.12885","metadata_source":"pith","pith_arxiv_id":"1810.04805","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","venue":"cs.CL","work_id":"ed240a10-5b19-406c-baa5-30803f465785","year":2018},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"cited_paper":"/paper/1810.04805","citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:eaefd20c7887973705473632888046188f8aeecf39d94b9cf90ec21931cd3824","observation_id":"5ad86f32-7955-4db5-89af-1ec1fe8cfd93","resolution":{"observed_at":"2026-05-17T20:22:35.143046Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Xia, Mehdi S","venue":null,"work_id":"93b31b44-b63f-41fa-bf56-6326a65df57d","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:d517027866b0501fc31a9ac026cfcacb3f01f576abfc9a36607fabe9bc1fd425","observation_id":"7d4ecb7f-b028-4a26-a347-e016f26f4a55","resolution":{"observed_at":"2026-05-17T20:22:35.316545Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.13394","last_updated":"2025-10-24T02:45:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-23T09:22:36Z","title":"MME: A Comprehensive Evaluation Benchmark for Multimodal Large Language Models","version":5},"cited_work":{"arxiv_id":"2306.13394","doi":"10.48550/arxiv.2306.13394","metadata_source":"pith","pith_arxiv_id":"2306.13394","snapshot_observed_at":"2026-07-10T13:27:05.561984Z","title":"MME: A Comprehensive Evaluation Benchmark for Multimodal Large Language Models","venue":"cs.CV","work_id":"806d2e73-71b3-4d56-87e0-39d571cc15d6","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"cited_paper":"/paper/2306.13394","citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:f1fb3364173d7e53d9b4c8dc66741856df60963d50617822eab3cb322d53ee85","observation_id":"490064e9-e5ef-4873-b3e8-33713a887665","resolution":{"observed_at":"2026-05-17T20:22:35.053895Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-08T16:08:17.033703+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T16:08:17.033703+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.12681","last_updated":"2022-04-16T04:21:26Z","snapshot_observed_at":"2026-08-04T04:22:15.975890Z","submitted_at":"2021-11-24T18:31:20Z","title":"VIOLET : End-to-End Video-Language Transformers with Masked Visual-token Modeling","version":2},"cited_work":{"arxiv_id":"2111.12681","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2111.12681","snapshot_observed_at":"2026-07-03T00:57:30.204945Z","title":"Violet: End-to-end video-language transformers with masked visual-token modeling","venue":null,"work_id":"c7602525-8155-4a61-bd4e-abf06c26e6e7","year":2021},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"cited_paper":"/paper/2111.12681","citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:29e949b8b01dfd206556c368d446b08e06ac1fcb4d711c33cea30b48253797e9","observation_id":"067cc1fb-e7f2-489f-a50a-ad1df38a4057","resolution":{"observed_at":"2026-05-17T20:22:35.059332Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mist : Multi-modal iterative spatial-temporal transformer for long-form video question answering","venue":null,"work_id":"f6deda80-7f97-4314-a2ac-370a78327d02","year":2022},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:eb25b710e0f296110ef7cd6d79f50582b99b77471ed6f5a6aa24a2f17c36a438","observation_id":"a851186a-6e04-4636-8230-9472ca2a0547","resolution":{"observed_at":"2026-05-17T20:22:35.319781Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gao, Chen Sun, Zhenheng Yang, and Ramakant Nevatia","venue":null,"work_id":"c5a3fa43-d77d-4300-a1f4-74f9cfb4c8d3","year":2017},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:ad8208f3f37888a16a4f5f30826380eb9207c068b1f0f5e924e80799d353e64f","observation_id":"3d21238a-6de4-4637-87ba-c1b3689434c6","resolution":{"observed_at":"2026-05-17T20:22:35.322872Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.04790","last_updated":"2023-06-13T13:31:12Z","snapshot_observed_at":"2026-07-06T15:24:38.226044Z","submitted_at":"2023-05-08T15:45:42Z","title":"MultiModal-GPT: A Vision and Language Model for Dialogue with Humans","version":3},"cited_work":{"arxiv_id":"2305.04790","doi":"10.48550/arxiv.2305.04790","metadata_source":"arxiv_reference","pith_arxiv_id":"2305.04790","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Multimodal-gpt: A vision and language model for dialogue with humans","venue":"arXiv (Cornell University)","work_id":"e5fb1f2e-4ed2-454f-87a3-9e9c40f8fa31","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"cited_paper":"/paper/2305.04790","citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:b374789e17e22bbcfb964f050c21aeee365740576e75ffbc4b115bf7edec3992","observation_id":"3f44ad43-9dd2-4b8c-8bd0-78657d63d7b0","resolution":{"observed_at":"2026-05-17T20:22:35.127729Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"something something","venue":null,"work_id":"ee92a1c8-c717-4d7d-9699-098f64744753","year":2017},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:12f1bf78059e6f11d939e1a571ab82e4835e26766af87796fd4ba12f21c37bf2","observation_id":"a7a0662b-e956-418c-a4ac-b0879b911ced","resolution":{"observed_at":"2026-05-17T20:22:35.325601Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Making the v in vqa matter: El- evating the role of image understanding in visual question answering","venue":null,"work_id":"a5a8ab0b-e8c4-4d64-bd8f-b1dde33f379d","year":2017},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:324ad4ff63b2af88178ceb80d6a129bfee410451fc2650b9c23b8ea1cc497784","observation_id":"6c708d15-c26e-48a9-8c76-ea2581ac6a17","resolution":{"observed_at":"2026-05-17T20:22:35.328577Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ramakrishnan, Fiona Ryan, Jayant Sharma, Michael Wray, Mengmeng Xu, Eric Z","venue":null,"work_id":"f757f2b0-d463-4697-8a6f-387fa7459a2f","year":2022},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:76188819825a9abd292f72e58b0ad00c99d31fd6ac0956b1ad2d187e55616ee1","observation_id":"3b2ce860-54ea-43a8-ab4b-40115579d15f","resolution":{"observed_at":"2026-05-17T20:22:35.332073Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Edward Hu, Yelong Shen, Phillip Wallis, Zeyuan Allen- Zhu, Yuanzhi Li, Shean Wang, and Weizhu Chen","venue":null,"work_id":"32bd7303-777a-4bc2-a748-d9cfb499eb9a","year":null},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:df8f3fc91e113c424ba24912cdf21832cc9c1477cc6f20120f8a4b0ac15071a1","observation_id":"066cf6d1-4f64-4f50-bc53-d3990f8098b4","resolution":{"observed_at":"2026-05-17T20:22:35.336385Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.14045","last_updated":"2023-03-01T11:04:51Z","snapshot_observed_at":"2026-08-08T05:52:17.343895Z","submitted_at":"2023-02-27T18:55:27Z","title":"Language Is Not All You Need: Aligning Perception with Language Models","version":2},"cited_work":{"arxiv_id":"2302.14045","doi":null,"metadata_source":"pith","pith_arxiv_id":"2302.14045","snapshot_observed_at":"2026-07-04T19:20:06.296460Z","title":"Language Is Not All You Need: Aligning Perception with Language Models","venue":"cs.CL","work_id":"2a1e0563-79f5-4521-8293-b8b1aebf7cee","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"cited_paper":"/paper/2302.14045","citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:28115315c55b51fe77bc7c449bab716724f0813b87f18a0936e70365bdf27972","observation_id":"70db8531-0909-409d-bbf5-548bb0ad9086","resolution":{"observed_at":"2026-05-17T20:22:35.063626Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Hudson and Christopher D","venue":null,"work_id":"235b4fbc-eec4-4f02-858c-b5d1f99a00a0","year":2019},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:abc458b9f75ba2a235a83f6cc85143f9cf30bb39b68a09045682d4416f14c2f2","observation_id":"ef3c492b-4165-4354-892d-41ee5ca3db79","resolution":{"observed_at":"2026-05-17T20:22:35.340231Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Jang, Yale Song, Youngjae Yu, Youngjin Kim, and Gun- hee Kim","venue":null,"work_id":"96041bd1-3e60-4a7a-9b20-01f2ef6aa38a","year":2017},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:988d5f1965a7051b792960137a095109364630483eeb61f21009ba987af0d739","observation_id":"d0946226-b6b3-4a07-91db-745b0bc46d52","resolution":{"observed_at":"2026-05-17T20:22:35.344812Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06825","last_updated":"2023-10-10T17:54:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-10T17:54:58Z","title":"Mistral 7B","version":1},"cited_work":{"arxiv_id":"2310.06825","doi":"10.48550/arxiv.2310.06825","metadata_source":"pith","pith_arxiv_id":"2310.06825","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mistral 7B","venue":"cs.CL","work_id":"eb5e1305-ad11-4875-ad8d-ad8b8f697599","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"cited_paper":"/paper/2310.06825","citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:6d1f93a5b276f3fc536ab4abef4e5bab0b158916afb316f0c6e47ce66ab47811","observation_id":"cd232c85-9449-46ce-a234-e511dc54e46c","resolution":{"observed_at":"2026-05-17T20:22:35.103462Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-10T22:08:12.954417+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-10T22:08:12.954417+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Lawrence Zitnick, and Ross B","venue":null,"work_id":"209038cc-cfb5-41ed-9d82-60b0f905c9a4","year":2017},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:110c774e5d03379011dab54af2ec4eb95d25862e6523a4fe2ce799b87468f523","observation_id":"b7ee2ce7-d33a-4b24-83f9-95e86febe9e3","resolution":{"observed_at":"2026-05-17T20:22:35.348342Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1705.06950","last_updated":"2017-05-19T12:07:01Z","snapshot_observed_at":"2026-08-08T17:46:50.107463Z","submitted_at":"2017-05-19T12:07:01Z","title":"The Kinetics Human Action Video Dataset","version":1},"cited_work":{"arxiv_id":"1705.06950","doi":null,"metadata_source":"pith","pith_arxiv_id":"1705.06950","snapshot_observed_at":"2026-07-09T11:16:11.423695Z","title":"The Kinetics Human Action Video Dataset","venue":"cs.CV","work_id":"c8a3de61-cfd3-4aeb-bcf7-a0372c015748","year":2017},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"cited_paper":"/paper/1705.06950","citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:1d0dada1f0be5fa18269f93147fba66541cd4e60bd9ca477f443ed1ee84c3e89","observation_id":"c6867e38-db18-4f71-8bd6-8a4ee3a4dd8f","resolution":{"observed_at":"2026-05-17T20:22:35.131390Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Beyond the nav-graph: Vision-and- language navigation in continuous environments","venue":null,"work_id":"4e0da732-83a2-4abb-aead-d1aabbdbe203","year":null},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:bb8371c26b87c130103576adc0c8a32e5080e5d98998a8f66f2ac84f01a4ea35","observation_id":"ae08a2f6-519e-420e-8d3b-b0d991e68af2","resolution":{"observed_at":"2026-05-17T20:22:35.352190Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A hierarchical approach for generating descriptive image paragraphs","venue":null,"work_id":"3ee38184-5023-43c0-a2e2-eb98aef02dbc","year":2017},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:b8bf704707a688865d9f2596abb3202e3715d2274112e67b189e75bd5643c4f0","observation_id":"bcbeb098-4575-4909-bd39-833c1a54f38a","resolution":{"observed_at":"2026-05-17T20:22:35.355843Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Visual genome: Connecting language and vision using crowdsourced dense image annotations","venue":null,"work_id":"0dce476d-033c-46e4-aab8-72adf9ad13de","year":2017},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:3fb8ec1d766427fe57c5f5620330fb1ac6afb2cd59ceda67c835dc2143b7cc9f","observation_id":"a695e460-0da8-4aae-a0f8-ac4b3f3eb42d","resolution":{"observed_at":"2026-05-17T20:22:35.359280Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"0d1e5f3e-3752-4b65-8eee-e9340a976960","year":2018},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:5189e62ebee5ea0ae8499626bac9c59cd842bb648687b280585409dda898c980","observation_id":"ca74c5b5-1186-4cce-a794-1f3d61e0311c","resolution":{"observed_at":"2026-05-17T20:22:35.362332Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Moreno, and Jes ´us Lov´on-Melgarejo","venue":null,"work_id":"cb88684f-2646-455e-8670-33407911798a","year":null},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:f7c2831b85b04c66aa00e2010b94368025e7a68051ac19a93bff09d3bd9c28bf","observation_id":"690b9d0a-9d42-4bac-af25-04d8b6a2d0af","resolution":{"observed_at":"2026-05-17T20:22:35.365833Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.16125","last_updated":"2023-08-02T08:02:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-30T04:25:16Z","title":"SEED-Bench: Benchmarking Multimodal LLMs with Generative Comprehension","version":2},"cited_work":{"arxiv_id":"2307.16125","doi":"10.48550/arxiv.2307.16125","metadata_source":"pith","pith_arxiv_id":"2307.16125","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SEED-Bench: Benchmarking Multimodal LLMs with Generative Comprehension","venue":"cs.CL","work_id":"23881ff0-b851-474c-8712-90744cc07a3a","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"cited_paper":"/paper/2307.16125","citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:a5ee4db7ec457c517486a30eebd524ab10982359f4446ae22943485ddb6c0b4f","observation_id":"30eeb6ab-3cc1-4e62-a226-dd766552f830","resolution":{"observed_at":"2026-05-17T20:22:35.068109Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.03726","last_updated":"2025-07-28T05:33:36Z","snapshot_observed_at":"2026-07-06T15:23:52.761222Z","submitted_at":"2023-05-05T17:59:46Z","title":"Otter: A Multi-Modal Model with In-Context Instruction Tuning","version":2},"cited_work":{"arxiv_id":"2305.03726","doi":"10.48550/arxiv.2305.03726","metadata_source":"pith","pith_arxiv_id":"2305.03726","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Otter: A Multi-Modal Model with In-Context Instruction Tuning","venue":"cs.CV","work_id":"33cb3a7a-6091-48db-a246-802bbb055f43","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"cited_paper":"/paper/2305.03726","citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:f53863d007e1f91e0c06055e82d09c440e2a1b6d25b7b230209da062a7973440","observation_id":"baca8caa-4445-42e7-8a03-44098fc3dba0","resolution":{"observed_at":"2026-05-17T20:22:35.072447Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"7d13f418-2ed6-414a-9f61-69dfac4982ab","year":2022},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:c6e1e8e81050b479ee22e382a18b4e36712ff80183144e5ff35d0fffeb0cae2a","observation_id":"14800046-ca1c-40ce-8666-66ef67006f86","resolution":{"observed_at":"2026-05-17T20:22:35.368773Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Inten- tqa: Context-aware video intent reasoning","venue":null,"work_id":"6f540966-fad1-4599-9bc8-54d7cdc42597","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:6b5c5eb04c5970bae3d8f2d75e39c30878ceed94d013ada18bc866e6c9903534","observation_id":"fd1694ba-3425-4fe8-b8d6-7a3d352f70be","resolution":{"observed_at":"2026-05-17T20:22:35.371437Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2211.09552","last_updated":"2022-11-17T14:17:40Z","snapshot_observed_at":"2026-08-07T04:39:25.258902Z","submitted_at":"2022-11-17T14:17:40Z","title":"UniFormerV2: Spatiotemporal Learning by Arming Image ViTs with Video UniFormer","version":1},"cited_work":{"arxiv_id":"2211.09552","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2211.09552","snapshot_observed_at":"2026-07-03T19:08:49.809225Z","title":"Uniformerv2: Spatiotemporal learning by arming image vits with video uniformer","venue":null,"work_id":"cb4c2612-0c2a-4ab9-9dde-7031260274d1","year":2022},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"cited_paper":"/paper/2211.09552","citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:fafe0a1e04379b3019e31374a54c8130fddce79f8fba50ca6c3f94f48c671965","observation_id":"eb5d0769-acbf-4ce3-9e46-8105f8c30d1f","resolution":{"observed_at":"2026-05-17T20:22:35.091018Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:3166837eefd6a7b730dd75d3619950b28683abb5cc4e38c53c8d3598540d22ea","observation_id":"884e9606-2235-40c7-b760-7076483c38d2","resolution":{"observed_at":"2026-05-17T20:22:35.099555Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Unmasked teacher: Towards training-efficient video foundation models","venue":null,"work_id":"ae69b7c0-189f-4db7-9a71-722cf439d525","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:1400b52c655cd03677b303ec42f287bb86fb71b58734a7779a3b909380f5d05e","observation_id":"5c253939-cc34-4738-ba97-ff8401600cc5","resolution":{"observed_at":"2026-05-17T20:22:35.374276Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.04387","last_updated":"2023-06-08T13:44:24Z","snapshot_observed_at":"2026-08-10T20:41:01.172916Z","submitted_at":"2023-06-07T12:35:37Z","title":"M$^3$IT: A Large-Scale Dataset towards Multi-Modal Multilingual Instruction Tuning","version":2},"cited_work":{"arxiv_id":"2306.04387","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.04387","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"M3it: A large-scale dataset towards multi-modal multilingual instruction tun- ing","venue":null,"work_id":"c402e58a-ad70-4eb6-8031-d70d3835ceab","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"cited_paper":"/paper/2306.04387","citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:7c6ed9ad133c3cc87957e9bf36b5ee695f49ac0685f2b2c9f537a9285e727fa6","observation_id":"ffad8371-cabe-4fd4-b398-09404bcf7c7a","resolution":{"observed_at":"2026-05-17T20:22:35.115909Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.10355","last_updated":"2023-10-26T02:52:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-17T16:34:01Z","title":"Evaluating Object Hallucination in Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":"2305.10355","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.10355","snapshot_observed_at":"2026-07-10T11:37:03.198858Z","title":"Evaluating Object Hallucination in Large Vision-Language Models","venue":"cs.CV","work_id":"66d8ac3e-c134-4995-b528-550afa17586f","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"cited_paper":"/paper/2305.10355","citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:781c58235c13614da3ba83e3aadfcac34a495b0badda8257d8e7dc3058998deb","observation_id":"15f2eb10-6ccb-41ef-bc89-8b39ce753c70","resolution":{"observed_at":"2026-05-17T20:22:35.119842Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Microsoft coco: Common objects in context","venue":null,"work_id":"67d48f28-533b-4e9c-a0ca-f80b99553418","year":2014},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:0fd9e2ece16e46757d0b6f21b42c03407454c682680b07c6d5509150051f3b3c","observation_id":"ac434c46-3150-4eed-9857-dcdd0b704875","resolution":{"observed_at":"2026-05-17T20:22:35.376770Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Visual instruction tuning","venue":null,"work_id":"96e6e514-fb82-400d-b018-f16b2fcd6d91","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:df76961d721df52c342b0b24eee97cac08be85b713bc55f97ca96b87484697d8","observation_id":"60236d7c-dc73-4f1e-8d74-fd71425a3de0","resolution":{"observed_at":"2026-05-17T20:22:35.379131Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ntu rgb+d 120: A large-scale benchmark for 3d human activity understand- ing","venue":null,"work_id":"30711739-d73f-405c-8937-11b43f7f2e41","year":2020},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:ba0c7df735820267a86726bfe364ff9ffcf54d2c15ebc003180e3ff795133ab9","observation_id":"ff17c2e8-d1ca-4029-946a-8a09807addf5","resolution":{"observed_at":"2026-05-17T20:22:35.381831Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.06281","last_updated":"2024-08-20T03:56:03Z","snapshot_observed_at":"2026-07-06T15:53:19.485466Z","submitted_at":"2023-07-12T16:23:09Z","title":"MMBench: Is Your Multi-modal Model an All-around Player?","version":5},"cited_work":{"arxiv_id":"2307.06281","doi":"10.48550/arxiv.2307.06281","metadata_source":"pith","pith_arxiv_id":"2307.06281","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMBench: Is Your Multi-modal Model an All-around Player?","venue":"cs.CV","work_id":"3b44943d-0f15-4228-9ac3-0e376f4f9ada","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"cited_paper":"/paper/2307.06281","citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:7316881b9b7d8072d903dc9c16e74707b77672fd65d9719f2d39c7517b711500","observation_id":"5d779c7c-2c89-45b0-aa56-dc03a5e0ff35","resolution":{"observed_at":"2026-05-17T20:22:35.147202Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.07207","last_updated":"2025-03-17T13:51:51Z","snapshot_observed_at":"2026-08-09T19:40:57.491981Z","submitted_at":"2023-06-12T16:11:10Z","title":"Valley: Video Assistant with Large Language model Enhanced abilitY","version":3},"cited_work":{"arxiv_id":"2306.07207","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.07207","snapshot_observed_at":"2026-07-04T19:20:06.409899Z","title":"Valley: Video assistant with large language model enhanced ability.arXiv preprint arXiv:2306.07207","venue":null,"work_id":"fdbffcd9-bed8-44ab-9171-37dea2a6f095","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"cited_paper":"/paper/2306.07207","citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:c46e30aea933926708c8b2b3e5eb319d3323de5f07d14da7aa81121704e8059e","observation_id":"f560f83b-f72e-4b21-9b3f-27f69a86c5a5","resolution":{"observed_at":"2026-05-17T20:22:35.014045Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05424","last_updated":"2024-06-10T01:36:53Z","snapshot_observed_at":"2026-07-06T15:40:24.127663Z","submitted_at":"2023-06-08T17:59:56Z","title":"Video-ChatGPT: Towards Detailed Video Understanding via Large Vision and Language Models","version":2},"cited_work":{"arxiv_id":"2306.05424","doi":"10.48550/arxiv.2306.05424","metadata_source":"pith","pith_arxiv_id":"2306.05424","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Video-ChatGPT: Towards Detailed Video Understanding via Large Vision and Language Models","venue":"cs.CV","work_id":"51f627f4-8fae-4882-a3e9-abdf932ef27b","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"cited_paper":"/paper/2306.05424","citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:efa1d7d4540c2d48c386536b16899ca95cc8b4e5b5b558945ca49d3b3dad3fff","observation_id":"706d5a7a-6d15-4eed-a84f-052294720fe6","resolution":{"observed_at":"2026-05-17T20:22:35.030804Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09126","last_updated":"2023-08-17T17:59:59Z","snapshot_observed_at":"2026-08-10T19:25:16.527226Z","submitted_at":"2023-08-17T17:59:59Z","title":"EgoSchema: A Diagnostic Benchmark for Very Long-form Video Language Understanding","version":1},"cited_work":{"arxiv_id":"2308.09126","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09126","snapshot_observed_at":"2026-07-04T16:39:57.306434Z","title":"Egoschema: A diagnostic benchmark for very long-form video language understanding","venue":null,"work_id":"f35031d8-b7b8-43e1-9226-b435af540d6a","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"cited_paper":"/paper/2308.09126","citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:d5fce640c675542f0d4750d470ef9d5858aa2ad312e6fc186955f7749e0d708e","observation_id":"64afd474-cd84-470f-a8c7-086f8848c8d4","resolution":{"observed_at":"2026-05-17T20:22:35.036418Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ok-vqa: A visual question answering benchmark requiring external knowledge","venue":null,"work_id":"089333a3-a25f-458f-9839-46b4e3c5a0a5","year":2019},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:5dffd6ae32f489f4b55d0989fd7752d02ab0abb8d1a1100f8159a648ea26f259","observation_id":"608c3e73-431a-4faf-a295-280d692164e7","resolution":{"observed_at":"2026-05-17T20:22:35.384389Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Manmatha, and C","venue":null,"work_id":"d4ef1362-3dd8-449f-b6a4-f1f4f013d3d5","year":2021},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:0c1382fb0d3a858cc3b7a57ae0d47dd875d8185c347d54ea0f6110710dd85cc2","observation_id":"c5f0ace5-ded8-4df3-9488-63f134d96c0e","resolution":{"observed_at":"2026-05-17T20:22:35.387189Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ocr-vqa: Visual question answering by reading text in images","venue":null,"work_id":"fa55f17f-10e3-478c-b2cb-96f576719f0f","year":2019},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:ef09ea9c3537d77ee6abf343a87b1973e7eed0dbae9b60821aed01cfcd7b977b","observation_id":"8d385301-dd3a-4c13-a487-c90ce82e7b19","resolution":{"observed_at":"2026-05-17T20:22:35.171192Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Spoken moments: Learning joint audio-visual representations from video de- scriptions","venue":null,"work_id":"6c601c9c-e97e-45b9-a683-45bc6d2f07e7","year":2021},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:9a99b745b4b3e86aae08fc224dcca3ae0b3cd636421caeb190cc83b6a512e43b","observation_id":"90ea304d-a46d-4428-ad42-b05a0c1866ab","resolution":{"observed_at":"2026-05-17T20:22:35.175006Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Brown, Quanfu Fan, Dan Gutfreund, Carl V ondrick, and Aude Oliva","venue":null,"work_id":"0ddcb4f0-e8c5-4da3-b7d3-3d266dbe601d","year":2020},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:9f545bfa06a66fbcde82f782c0f76a88e60cc155aab7ad01c9e68d3b29442518","observation_id":"0faf0da3-d353-42d0-a1a7-57b4ccf3ecfe","resolution":{"observed_at":"2026-05-17T20:22:35.178454Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"644a3abd-72e8-4d67-99f2-f95a02510535","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:51b014c34b0f887ba36ec3ac700411d15855707123afb4552664a77029a2d1c3","observation_id":"82c7baf6-c260-47c2-9721-59e293ce147e","resolution":{"observed_at":"2026-05-17T20:22:35.181408Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gpt-4v(ision) system card","venue":null,"work_id":"200b4a2d-210b-4767-92c5-9e7b190e2149","year":null},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:199a40e17d49307f9fd62bb542f5e28c1531f38edb98ddb8889152e8d04fc8b2","observation_id":"f9097022-6d10-46cc-94b3-4f05fff240a0","resolution":{"observed_at":"2026-05-17T20:22:35.184444Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Im2text: Describing images using 1 million captioned pho- tographs","venue":null,"work_id":"9b253296-c178-474d-92fd-2c0f8f205a09","year":2011},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:6b9bfeab6215930256012831c638cbee05cca81b1678f87720afe6ac2b103db0","observation_id":"b73ffd91-2fd3-46d4-99b8-b47f03135d92","resolution":{"observed_at":"2026-05-17T20:22:35.187488Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Koster, Junlin Zhang, Stephanie, Winkler, Yusuf Aytar, Si- mon Osindero, Dima Damen, Andrew Zisserman, and Jo˜ao Carreira","venue":null,"work_id":"920baede-424e-496f-ba9b-f9becf0cb167","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:a9ab46b52ca5ce2da0139b5f7e19bdb398c75b2ef87bc08f101fdfeff49e1f76","observation_id":"6499d076-800b-459b-8d4d-479ba6c6d609","resolution":{"observed_at":"2026-05-17T20:22:35.190584Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Flickr30k entities: Collecting region-to-phrase corre- spondences for richer image-to-sentence models","venue":null,"work_id":"bcfbcedc-6258-42d8-ad11-87076c23cae3","year":null},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:66e95c224e3d19c45981cc9e031cdc05fd1ee5e8cbbefc216f764d34265270df","observation_id":"a39c1eec-c583-4a72-9465-f206b833dd48","resolution":{"observed_at":"2026-05-17T20:22:35.194198Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Exploring the limits of transfer learning with a unified text-to-text transformer","venue":null,"work_id":"354109c3-3a5d-4ba7-9cc0-e53127eae124","year":2020},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:6d5170526578bfb10c2588a1d917c5687d303ad02b95a348d5b83060a15f204d","observation_id":"d000b56b-1921-41be-9edb-3d2764562e6a","resolution":{"observed_at":"2026-05-17T20:22:35.197586Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A-okvqa: A benchmark for visual question answering using world knowledge","venue":null,"work_id":"eadabe7f-666a-43dc-a055-5423cd9e3264","year":2022},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:75dca7b96f198a84a8fae4a1ed893e6a8c149e8cdc50e5d68d1f3ca91d695d0f","observation_id":"de097274-2a17-4332-8f47-9759e0228f82","resolution":{"observed_at":"2026-05-17T20:22:35.200731Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Conceptual captions: A cleaned, hypernymed, im- age alt-text dataset for automatic image captioning","venue":null,"work_id":"f9eacc61-38c6-4947-92f7-77a6f73bd2dc","year":null},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:34c889b93f32a4732a138c784f735a4ddf35b37e45353a8db933f76c3950c7ba","observation_id":"e6902d31-2215-4dc4-ac90-278be090f162","resolution":{"observed_at":"2026-05-17T20:22:35.204226Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textcaps: a dataset for image caption- ing with reading comprehension","venue":null,"work_id":"b8594278-6288-410c-9fbf-8151074723cb","year":2020},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:9542234b1a7d90321ef66edba963e345f5d834d4ebb462c60eef83ca87cfdf1d","observation_id":"837d356b-1914-4a8e-aef3-bf846d1da07f","resolution":{"observed_at":"2026-05-17T20:22:35.207541Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Towards vqa models that can read","venue":null,"work_id":"2cbd86ed-fba5-4b1c-9b48-1fe75d6bf2a8","year":null},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:0d78342f0e11546a4c77003e4c44b324360968aee667c9881231187195f9d735","observation_id":"a7e08426-b268-4fe6-ac78-bbcd4b051f4c","resolution":{"observed_at":"2026-05-17T20:22:35.210930Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.15389","last_updated":"2023-03-27T17:02:21Z","snapshot_observed_at":"2026-07-06T15:08:34.018146Z","submitted_at":"2023-03-27T17:02:21Z","title":"EVA-CLIP: Improved Training Techniques for CLIP at Scale","version":1},"cited_work":{"arxiv_id":"2303.15389","doi":"10.48550/arxiv.2303.15389","metadata_source":"pith","pith_arxiv_id":"2303.15389","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"EVA-CLIP: Improved Training Techniques for CLIP at Scale","venue":"cs.CV","work_id":"0c16c250-fd0f-446a-bbb0-ea8dd0ba5ccd","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"cited_paper":"/paper/2303.15389","citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:2771c9235eddd6fa896dc9ee1c2c2fe1acd6d2a57fecfb275f0cdc1a4828f315","observation_id":"0c9fbff2-1c1d-46ed-8d07-0a09f3a29daf","resolution":{"observed_at":"2026-05-17T20:22:35.139025Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vi- sualmrc: Machine reading comprehension on document im- ages","venue":null,"work_id":"60f49a6c-06b2-4370-9db5-0b79f880f9dd","year":2021},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:3e553ec545f74bbbfaece07d0081e6fc54c00cd44dc7ca2bda2a33a971d588f1","observation_id":"3c66ae51-ea23-4424-a939-b9fd94be8923","resolution":{"observed_at":"2026-05-17T20:22:35.213687Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Internlm: A multilingual language model with progressively enhanced capabilities","venue":null,"work_id":"9dbebdd3-d18a-47e9-9f3a-48a76221a518","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:68b9a40aa091752b90db1c3adafd534474dff2732265366188abba721ffa1bc3","observation_id":"ab0672ca-c800-4850-9011-4c96415d878e","resolution":{"observed_at":"2026-05-17T20:22:35.216815Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vicuna: An open-source chatbot impress- ing gpt-4 with 90% chatgpt quality","venue":null,"work_id":"61c2d321-6beb-465d-ac3e-d2e2fe3f4fe8","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:c4dd9fc4b7364c70f47b7d20285683cedac4e2238540fe0103090c505e53e9e2","observation_id":"8b001f02-1935-40b6-b1e9-07b8d5b49dc0","resolution":{"observed_at":"2026-05-17T20:22:35.220502Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":"2302.13971","doi":"10.48550/arxiv.2302.13971","metadata_source":"pith","pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LLaMA: Open and Efficient Foundation Language Models","venue":"cs.CL","work_id":"c018fc23-6f3f-4035-9d02-28a2173b2b9d","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:efebdf083c4930ae348c48dd629df91c5b6cf39bde3dabf10cc3bdcaa5f352e7","observation_id":"d5660cc0-b6fb-4c52-a400-f2744745d926","resolution":{"observed_at":"2026-05-17T20:22:35.021031Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-08T16:08:17.350515+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T16:08:17.350515+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":"2307.09288","doi":"10.24963/ijcai.2025/706","metadata_source":"pith","pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","venue":"cs.CL","work_id":"68a5177f-d644-44c1-bd4f-4e5278c22f5d","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:42dafe4f61787ba93e3c8ed74d278d64771ec1a04096d0ffa48b6028f7f1a104","observation_id":"d68bc025-cdea-4ebb-a0f5-38e38cf1201c","resolution":{"observed_at":"2026-05-17T20:22:35.025946Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"All in one: Exploring unified video-language pre-training","venue":null,"work_id":"b1a0ae18-e795-428b-b411-398f6d5e62e7","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:f666d13c640e4020e7a289b6135100d5223b5367d61be00b256e815f61f9e3db","observation_id":"6e895c53-b0ff-4461-8984-0ecf93c6461a","resolution":{"observed_at":"2026-05-17T20:22:35.224621Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Temporal segment networks: Towards good practices for deep action recogni- tion","venue":null,"work_id":"35dd5bf1-3df4-464b-8e31-bf2ec30ca411","year":2016},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:57dbdfde718ba7c6544f66e32c049a596fbeb068e3d3db092a557484c1281fcc","observation_id":"e73b2057-7119-4098-b954-52adf08ff6bc","resolution":{"observed_at":"2026-05-17T20:22:35.228698Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Videomae v2: Scaling video masked autoencoders with dual masking","venue":null,"work_id":"738ffcde-96d2-4daf-a294-353aa56cabe6","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:17e06a8093ad18eb2052ee77c5bf6e7187ccbdc8247227ac798cdaa57169b5f1","observation_id":"9d3a4a30-1b12-4082-99cf-9ce870ff4ee8","resolution":{"observed_at":"2026-05-17T20:22:35.231827Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.03191","last_updated":"2022-12-07T12:20:55Z","snapshot_observed_at":"2026-07-06T14:27:34.639236Z","submitted_at":"2022-12-06T18:09:49Z","title":"InternVideo: General Video Foundation Models via Generative and Discriminative Learning","version":2},"cited_work":{"arxiv_id":"2212.03191","doi":"10.48550/arxiv.2212.03191","metadata_source":"pith","pith_arxiv_id":"2212.03191","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo: General Video Foundation Models via Generative and Discriminative Learning","venue":"cs.CV","work_id":"780aaeee-ac26-46b1-b6ff-64a7a624e694","year":2022},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"cited_paper":"/paper/2212.03191","citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:dc09adf421a127b178cd6ed9be941ed924ab3c53244faba5e38d23fcb47130b6","observation_id":"107448f9-83d7-444d-b0ef-af0b68ae0758","resolution":{"observed_at":"2026-05-17T20:22:35.048363Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"260220e2-236f-415c-9de3-a52c6a6b809c","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:6be6cc51344b6116e577191382b7cda22545a014eff05c79f06a991c4549a28d","observation_id":"4ebd95a4-819d-48a7-8632-f753c3b93a26","resolution":{"observed_at":"2026-05-17T20:22:35.235879Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Pax- ion: Patching action knowledge in video-language founda- tion models","venue":null,"work_id":"e4efd665-2318-4405-b161-e2f052afccae","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:eb043b8d4fb595e08aaf907652adc388cbdc3587e19e875fde8dd06722380468","observation_id":"d921c129-0c8c-43f4-81b4-a572d9209444","resolution":{"observed_at":"2026-05-17T20:22:35.238971Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Dai, and Quoc V","venue":null,"work_id":"46406fe4-4eeb-40bf-8699-3ffe0d98e85f","year":2021},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:dad02c12a6179a9dc51aa629ed627465e20183921b308ec55554eed5b9d0a313","observation_id":"02f8d3ea-133d-4dc4-a69a-d21891a158ad","resolution":{"observed_at":"2026-05-17T20:22:35.241492Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Chi, Quoc V Le, and Denny Zhou","venue":null,"work_id":"d853e75c-a8e8-44b0-a928-e80ba18333b4","year":2022},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:1c058a9e8a2e7966a56170cfe29803076624780158a6bc56042288b26089a68a","observation_id":"2dabd89e-dac9-49a6-87b8-96e0ed1551bf","resolution":{"observed_at":"2026-05-17T20:22:35.244201Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Tenen- baum, and Chuang Gan","venue":null,"work_id":"442ce091-b7fa-41ab-83de-da303fe13062","year":2021},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:84c4882059363f274e14bd84cedf6ce637330d5fcab475381de89b9454ee7501","observation_id":"2b684e64-4a4d-45f2-b4df-d9b2da82d432","resolution":{"observed_at":"2026-05-17T20:22:35.247221Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.03347","last_updated":"2023-05-05T08:00:14Z","snapshot_observed_at":"2026-08-11T10:38:46.382315Z","submitted_at":"2023-05-05T08:00:14Z","title":"A Large Cross-Modal Video Retrieval Dataset with Reading Comprehension","version":1},"cited_work":{"arxiv_id":"2305.03347","doi":"10.48550/arxiv.2305.03347","metadata_source":"arxiv_reference","pith_arxiv_id":"2305.03347","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A large cross- modal video retrieval dataset with reading comprehension","venue":"arXiv (Cornell University)","work_id":"e573b642-523e-496e-b0bd-65f57fa974f4","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"cited_paper":"/paper/2305.03347","citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:af0604d8c8a06ca6726c14aa1f1e7204dc2376cee75229f3b9cf0ac7b5de7010","observation_id":"1b30f322-f650-49d0-a240-162c68223b5d","resolution":{"observed_at":"2026-05-17T20:22:35.077258Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Next-qa: Next phase of question-answering to explaining temporal actions","venue":null,"work_id":"780e6dc6-8070-47d9-8489-ed8aacca1d57","year":2021},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:f5f77d91a0ed5eb827a7d4a736e04e4a8622d0a0227921f34b9360f7bc4b1903","observation_id":"7cd696e1-d891-48f1-bdbd-1e220a9dd4d9","resolution":{"observed_at":"2026-05-17T20:22:35.250280Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Video as conditional graph hierarchy for multi-granular question answering","venue":null,"work_id":"7b677015-3a2f-40b6-971d-363f7f08abe6","year":2022},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:57641afdb0c7ee0a0d287c73966a31dfb9a4d20a1233b52b6049744ae94f4766","observation_id":"7e7b6dab-6a20-4f87-b281-d19dcd8f2150","resolution":{"observed_at":"2026-05-17T20:22:35.253042Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Video graph transformer for video question answering","venue":null,"work_id":"a8268e50-3f0b-4482-a49e-d6ad928c6688","year":2022},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:c678636a15e680627b4e6cf6db6e57390539d76b285c8107fa6773d4269ad2a5","observation_id":"dccfa900-c152-417b-a495-35ffd433cc0a","resolution":{"observed_at":"2026-05-17T20:22:35.255929Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14899","last_updated":"2024-03-22T13:24:35Z","snapshot_observed_at":"2026-07-31T06:40:35.464819Z","submitted_at":"2023-06-26T17:59:55Z","title":"FunQA: Towards Surprising Video Comprehension","version":2},"cited_work":{"arxiv_id":"2306.14899","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.14899","snapshot_observed_at":"2026-07-04T06:39:37.649762Z","title":"Funqa: Towards surprising video comprehension","venue":null,"work_id":"fa8eaebb-a95a-4f34-80af-f000e5edc756","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"cited_paper":"/paper/2306.14899","citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:c5d42f4f7b7bbbe6910746a3e5d0d771be50a0091a01a9ae9514bde2889aacaa","observation_id":"b724392e-2dc2-40f1-99df-59cc4c16f0bd","resolution":{"observed_at":"2026-05-17T20:22:35.095848Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Video question answer- ing via gradually refined attention over appearance and mo- tion","venue":null,"work_id":"bd9e3354-76d5-4424-a609-badec5a6881d","year":2017},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:9384007cc5e79dc01ee76c12f1e333f5c423523ececdb7ebeee27d22bb44a7c3","observation_id":"0b2e3657-ec4f-4b5a-a812-8c5fb40b800d","resolution":{"observed_at":"2026-05-17T20:22:35.258556Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Msr-vtt: A large video description dataset for bridging video and language","venue":null,"work_id":"af35820e-8c7b-4da7-ae4f-57322fcc1f53","year":2016},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:4d50ba2633abcf6715597cc715419ad4fe251bc8988e666a5f0becfa8736f93c","observation_id":"aa04807c-dc3d-48f3-9d7e-940b3781dbb8","resolution":{"observed_at":"2026-05-17T20:22:35.261498Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.09265","last_updated":"2023-06-15T16:39:24Z","snapshot_observed_at":"2026-07-06T15:43:03.878575Z","submitted_at":"2023-06-15T16:39:24Z","title":"LVLM-eHub: A Comprehensive Evaluation Benchmark for Large Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2306.09265","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.09265","snapshot_observed_at":"2026-07-04T03:29:31.137857Z","title":"Lvlm-ehub: A comprehensive evaluation benchmark for large vision-language models","venue":null,"work_id":"fc8d0558-bbd9-4655-b4b5-84d23092ff30","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"cited_paper":"/paper/2306.09265","citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:51896c74b0a1b74197a3d5f16a8ae5caae60937cadd5df7e23998fbd1ad89f55","observation_id":"8315d76d-f51f-4b95-b08e-8b1953fce6aa","resolution":{"observed_at":"2026-05-17T20:22:35.107932Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Just ask: Learning to answer questions from millions of narrated videos","venue":null,"work_id":"45c3c383-4e88-412f-8eef-31d5727c6701","year":2021},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:8d20d1ec7725e9745950999aca91eb37793fec5b5ecf5a69cf197c2fc202fd73","observation_id":"b7637b85-bb7b-4170-97b1-1bfcc189b629","resolution":{"observed_at":"2026-05-17T20:22:35.284437Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Zero-shot video question answering via frozen bidirectional language models","venue":null,"work_id":"bf658258-cba4-4cc6-b714-5915ab817145","year":2022},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:af398161e72f948c2d7033f020755bb9e470b52a6bcb73bf20b2f3f00b569397","observation_id":"2a7ea0bb-f1e3-46f7-a52d-746ac6284090","resolution":{"observed_at":"2026-05-17T20:22:35.264671Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Hitea: Hierarchical temporal- aware video-language pre-training","venue":null,"work_id":"23965e2f-e73a-4404-bbc9-f55c1d5abe66","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:a76511333427066d42a5145c86519395f0434a356bc0e368a261c6085d7953c1","observation_id":"b2b566d4-c165-49bc-b3dd-c8336ae6afa0","resolution":{"observed_at":"2026-05-17T20:22:35.267646Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:5ef358121d7c1c9d953b0710426a3c0a891e9c2b54e397cedd7bd371721d9f22","observation_id":"54661be6-1b54-4002-b669-da2f5e91851d","resolution":{"observed_at":"2026-05-17T20:22:35.123593Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Tenenbaum","venue":null,"work_id":"dd900f38-f1fe-4629-aa17-0f91b241b5d6","year":2020},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:ce2e419e0797174035d1a7482f5c885a4f38015dfa8c1afa11175ad6099eb6fc","observation_id":"6b79f101-d438-4515-b2c7-c0540d3c4f1c","resolution":{"observed_at":"2026-05-17T20:22:35.270161Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Self-chained image-language model for video localization and question answering","venue":null,"work_id":"168bf830-a9c9-48f5-bc23-6aaedc3a4df4","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:4e00c847e83f83eb50375ce6e0c27d6875dc380c7cf33206ca07c9166470664e","observation_id":"88e9485e-5ce6-4160-b357-820ea91ae442","resolution":{"observed_at":"2026-05-17T20:22:35.272772Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.02490","last_updated":"2024-12-01T05:46:03Z","snapshot_observed_at":"2026-08-08T03:31:37.699253Z","submitted_at":"2023-08-04T17:59:47Z","title":"MM-Vet: Evaluating Large Multimodal Models for Integrated Capabilities","version":4},"cited_work":{"arxiv_id":"2308.02490","doi":null,"metadata_source":"pith","pith_arxiv_id":"2308.02490","snapshot_observed_at":"2026-07-10T09:37:00.827227Z","title":"MM-Vet: Evaluating Large Multimodal Models for Integrated Capabilities","venue":"cs.AI","work_id":"7f3bac41-a0a5-4a7a-bfd2-526b616db745","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"cited_paper":"/paper/2308.02490","citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:57cface33dd1af1baefe8ad75564d9b112b26ac041fc4bed9bc20aa68bafe67f","observation_id":"0c0e51cf-69ea-43a2-a439-32182c12aad7","resolution":{"observed_at":"2026-05-17T20:22:35.135251Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Activitynet-qa: A dataset for understanding complex web videos via question answering","venue":null,"work_id":"49aad0f6-219c-40cf-b48f-72ca463386e8","year":2019},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:8e4cbb39585dd17e7be6859b924b29beeefbb8d62e5f5c254e6220590e6b216b","observation_id":"f6d84558-72db-43b3-9251-f5187fdcd617","resolution":{"observed_at":"2026-05-17T20:22:35.275559Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Activitynet-qa: A dataset for understanding complex web videos via question answering","venue":null,"work_id":"6631d55a-7b93-459a-8d25-8048b45f18f7","year":2019},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:50966bfb441af865d4a4f51208b420376e628853bebb22cfc047c8a6b8e7bc4d","observation_id":"5fd0b994-56bd-42e3-ac4c-52e295b56df8","resolution":{"observed_at":"2026-05-17T20:22:35.278661Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Zhang, Yuxiao Dong, and Jie Tang","venue":null,"work_id":"9968c083-f7ba-43df-9f80-2eed2d26fd19","year":2022},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:0f0782f52d96957459fdde7620ba418aa40c22bf515bc2085153f5c99e18773b","observation_id":"86529275-f921-4e96-940d-fc3d046343d9","resolution":{"observed_at":"2026-05-17T20:22:35.281440Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","latest_version":4,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":6,"verified_exact":29,"verified_fuzzy":64},"total_outbound_references":104},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 100 of 104 outbound references and 51 inbound Pith citation observations for arXiv:2311.17005."}