{"as_of":"2026-08-11T13:08:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:ce19a56fe0f81c31d062e5542c8e0de24bffe79484643f88a19f6ff17336a806","coverage":[{"denominator":37,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":37,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-17T02:52:20.643070Z","state":"measured"},{"denominator":121,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":121,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":84,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":84,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:41:08.446932Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":1,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2504.06958","last_updated":"2025-11-11T08:30:00Z","snapshot_observed_at":"2026-08-02T02:31:33.589341Z","submitted_at":"2025-04-09T15:09:27Z","title":"VideoChat-R1: Enhancing Spatio-Temporal Perception via Reinforcement Fine-Tuning","version":5},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-15T20:56:07.247122Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2504.06958"},"observation_digest":"sha256:7120769de62b77e256e3fde5f39d40671f3b5d84efd5200308a274575d075d44","observation_id":"a0ccc2c1-0ac2-406c-8c64-f23499a54bad","resolution":{"observed_at":"2026-05-17T02:52:20.835573Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-07T15:41:08.446932Z","title":"Internvideo2","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:08.446932Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2505.14321"},"observation_digest":"sha256:216d15575a5a09422e1ca7d3c4a665bd3f91e2d0de2de229cda55e5356eab0fc","observation_id":"33c74869-a7ea-414b-91da-4bde70336a44","resolution":{"observed_at":"2026-08-07T15:41:08.446932Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-07T15:34:41.079838Z","title":"Internvideo2","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-10T08:26:12.719387Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:41.079838Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:53d497be4d44a6d0d535dc4493af845cb24a7c7a2a737f31ae1b88583870d311","observation_id":"a97207d3-7a08-4b04-b7be-0cd06df75772","resolution":{"observed_at":"2026-08-07T15:34:41.079838Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-07T15:20:59.072257Z","title":"Internvideo2","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15447","last_updated":"2025-05-21T12:29:40Z","snapshot_observed_at":"2026-08-11T01:33:01.559626Z","submitted_at":"2025-05-21T12:29:40Z","title":"ViaRL: Adaptive Temporal Grounding via Visual Iterated Amplification Reinforcement Learning","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T15:20:59.072257Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2505.15447"},"observation_digest":"sha256:de63e113d8ba957c9300fc9c2c3c47346dfc101ff2a3855a58f4789f2065fcfa","observation_id":"8ffb3ee0-50c0-4ea5-affe-48f2ee8e725d","resolution":{"observed_at":"2026-08-07T15:20:59.072257Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-07T15:09:01.009435Z","title":"Internvideo2.5: Empowering video mllms with long and rich context modeling","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.16175","last_updated":"2025-05-31T13:43:36Z","snapshot_observed_at":"2026-08-08T10:58:05.073646Z","submitted_at":"2025-05-22T03:26:50Z","title":"QuickVideo: Real-Time Long Video Understanding with System Algorithm Co-Design","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T15:09:01.009435Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2505.16175"},"observation_digest":"sha256:12f7365eedd2aaae2b1ffed31e07cfa0ebf0dd516d35b42cc9d44261b1d28898","observation_id":"3b4f82a9-424b-4afe-922d-3d11b78ce6a3","resolution":{"observed_at":"2026-08-07T15:09:01.009435Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2505.16933","last_updated":"2025-06-04T05:52:10Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-22T17:23:26Z","title":"LLaDA-V: Large Language Diffusion Models with Visual Instruction Tuning","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-17T03:46:06.074416Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2505.16933"},"observation_digest":"sha256:ceb805d3ab83184e7b66ef465c79e8721d8e09c934a0c26ebcf8d4c45e5efe9b","observation_id":"40cb993f-0961-4de2-82ad-085c8318420d","resolution":{"observed_at":"2026-05-17T03:46:06.399421Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-07T14:44:30.557505Z","title":"Internvideo2","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17807","last_updated":"2025-05-23T12:24:28Z","snapshot_observed_at":"2026-08-10T23:27:52.942878Z","submitted_at":"2025-05-23T12:24:28Z","title":"Temporal Consistency Constrained Transferable Adversarial Attacks with Background Mixup for Action Recognition","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T14:44:30.557505Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2505.17807"},"observation_digest":"sha256:979c85dd2b873e8a7be1683a5dba618d1d1dcf9410d809dfa58304999c71a4bc","observation_id":"0d7da37b-986e-4383-9fe7-4e56a55a6701","resolution":{"observed_at":"2026-08-07T14:44:30.557505Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-07T14:14:45.104101Z","title":"Internvideo2","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19650","last_updated":"2025-05-27T11:57:17Z","snapshot_observed_at":"2026-08-09T09:30:25.697403Z","submitted_at":"2025-05-26T08:09:44Z","title":"Modality Curation: Building Universal Embeddings for Advanced Multimodal Information Retrieval","version":2},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-07T14:14:45.104101Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2505.19650"},"observation_digest":"sha256:6974429b478ac58da9bae386630c3e3ca1de6fef06206b7b27d9e19db1126928","observation_id":"f54c8e76-bdc0-480b-9fc8-a497a0ec08de","resolution":{"observed_at":"2026-08-07T14:14:45.104101Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-07T14:09:13.309276Z","title":"Internvideo2","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19877","last_updated":"2025-05-26T12:05:16Z","snapshot_observed_at":"2026-08-10T09:10:24.017088Z","submitted_at":"2025-05-26T12:05:16Z","title":"Vad-R1: Towards Video Anomaly Reasoning via Perception-to-Cognition Chain-of-Thought","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-07T14:09:13.309276Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2505.19877"},"observation_digest":"sha256:21e3b625bfec548f3e863706ca227240482e4b3557c74a7884e8c0ca8d370918","observation_id":"01a233a0-30b5-42cc-a9b7-5a273bf9102d","resolution":{"observed_at":"2026-08-07T14:09:13.309276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-07T12:32:54.252300Z","title":"Internvideo2","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.24329","last_updated":"2025-07-31T03:03:18Z","snapshot_observed_at":"2026-08-09T06:10:22.489105Z","submitted_at":"2025-05-30T08:10:18Z","title":"DisTime: Distribution-based Time Representation for Video Large Language Models","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T12:32:54.252300Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2505.24329"},"observation_digest":"sha256:6fb9e14621bd26a123152cf599d6987d82419f4167457baf76eb4a16b5a81393","observation_id":"ee4104bc-0d5b-416d-9007-440cd0f92730","resolution":{"observed_at":"2026-08-07T12:32:54.252300Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2506.01844","last_updated":"2025-06-02T16:30:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-02T16:30:19Z","title":"SmolVLA: A Vision-Language-Action Model for Affordable and Efficient Robotics","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-11T21:22:36.902119Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2506.01844"},"observation_digest":"sha256:d5979f6ee1269b1c26d86b33d879329d7c4bdbf07acd48ed2cde6646418b5c06","observation_id":"f1ad3c69-2093-45df-9890-bc1709f523ce","resolution":{"observed_at":"2026-05-17T02:52:20.835573Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-07T11:35:47.149960Z","title":"Internvideo2","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.01908","last_updated":"2025-06-02T17:28:26Z","snapshot_observed_at":"2026-08-07T11:29:22.385642Z","submitted_at":"2025-06-02T17:28:26Z","title":"Reinforcement Learning Tuning for VideoLLMs: Reward Design and Data Efficiency","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T11:35:47.149960Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2506.01908"},"observation_digest":"sha256:b7c23b15934186ec666a6cc9dfe8a99300895ecdb1c8a4446ac1533515722083","observation_id":"c8498e34-385e-46d5-b571-f9cbbe09859c","resolution":{"observed_at":"2026-08-07T11:35:47.149960Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-07T10:51:42.973201Z","title":"Internvideo2","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.04216","last_updated":"2025-06-04T17:57:43Z","snapshot_observed_at":"2026-08-08T15:17:23.629862Z","submitted_at":"2025-06-04T17:57:43Z","title":"UNIC: Unified In-Context Video Editing","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T10:51:42.973201Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2506.04216"},"observation_digest":"sha256:001a160b32d6633c440d572b5a3d645d850de2c5c6b049933c90fa9bd7af925b","observation_id":"9e4c774c-c2f9-4a2b-8269-0bdcd2437833","resolution":{"observed_at":"2026-08-07T10:51:42.973201Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-07T10:27:05.618482Z","title":"Internvideo2.5: Empowering video mllms with long and rich context modeling.arXiv preprint arXiv:2501.12386, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.05328","last_updated":"2025-07-22T07:00:35Z","snapshot_observed_at":"2026-08-08T19:39:48.959316Z","submitted_at":"2025-06-05T17:58:33Z","title":"AV-Reasoner: Improving and Benchmarking Clue-Grounded Audio-Visual Counting for MLLMs","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T10:27:05.618482Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2506.05328"},"observation_digest":"sha256:ab3dee21f25732a7d0a630ceed1ce6d73a02661e076fc31c62c9fac120968fad","observation_id":"8cf9e3ab-02f8-44a7-a346-1c46941f6bbd","resolution":{"observed_at":"2026-08-07T10:27:05.618482Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-07T10:25:36.311280Z","title":"Internvideo2","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.05349","last_updated":"2025-06-24T07:36:56Z","snapshot_observed_at":"2026-08-07T23:48:12.277994Z","submitted_at":"2025-06-05T17:59:58Z","title":"VideoMathQA: Benchmarking Mathematical Reasoning via Multimodal Understanding in Videos","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T10:25:36.311280Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2506.05349"},"observation_digest":"sha256:dbaf8f75e71ef6194bf088ba716c04ecefd0d49965101a528cc472f04223eaa4","observation_id":"580bc404-e098-47ea-8eca-2769ff68229a","resolution":{"observed_at":"2026-08-07T10:25:36.311280Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-07T05:43:43.067246Z","title":"InternVideo2.5: Empowering video mllms with long and rich context modeling","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.07202","last_updated":"2025-06-08T15:52:38Z","snapshot_observed_at":"2026-08-11T11:34:16.404599Z","submitted_at":"2025-06-08T15:52:38Z","title":"Reasoning Multimodal Large Language Model: Data Contamination and Dynamic Evaluation","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T05:43:43.067246Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2506.07202"},"observation_digest":"sha256:910c080b702c4fec69270410ad74b36c71bef71c714c03db37535ec36f1c9267","observation_id":"fc628e12-ea92-4ca2-8b7b-a2916806fa5f","resolution":{"observed_at":"2026-08-07T05:43:43.067246Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-07T05:26:47.266447Z","title":"Internvideo2","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.07971","last_updated":"2025-06-09T17:45:18Z","snapshot_observed_at":"2026-08-09T13:06:51.941159Z","submitted_at":"2025-06-09T17:45:18Z","title":"CyberV: Cybernetics for Test-time Scaling in Video Understanding","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T05:26:47.266447Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2506.07971"},"observation_digest":"sha256:932ae87549989218750726c3ad763139406561f6646e2a1fc62f3470236ee409","observation_id":"61606932-3186-4d92-91fb-d4541afcc799","resolution":{"observed_at":"2026-08-07T05:26:47.266447Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-07T05:05:51.389244Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.08817","last_updated":"2025-06-12T15:51:33Z","snapshot_observed_at":"2026-08-10T09:10:24.911665Z","submitted_at":"2025-06-10T14:08:56Z","title":"Video-CoT: A Comprehensive Dataset for Spatiotemporal Understanding of Videos Based on Chain-of-Thought","version":3},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T05:05:51.389244Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2506.08817"},"observation_digest":"sha256:4c52862b41fb90e5a7c807b2b8270458aeac94d047709527949e397dbfb4c78a","observation_id":"dc595d87-8263-4001-ad95-c20fdf1fbb47","resolution":{"observed_at":"2026-08-07T05:05:51.389244Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-07T04:22:56.040389Z","title":"Internvideo2","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.10857","last_updated":"2025-08-04T09:11:48Z","snapshot_observed_at":"2026-08-08T13:30:57.844783Z","submitted_at":"2025-06-12T16:17:17Z","title":"VRBench: A Benchmark for Multi-Step Reasoning in Long Narrative Videos","version":2},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-07T04:22:56.040389Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2506.10857"},"observation_digest":"sha256:bb6efd75e8067fbdac6e0235059417bd1128e6e95e99cd261dac8edfa9ab5cc4","observation_id":"a8c1bebf-3b71-4951-8ed2-20c1839bf3c5","resolution":{"observed_at":"2026-08-07T04:22:56.040389Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-07T00:34:36.092629Z","title":"Internvideo2","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.13654","last_updated":"2025-06-16T16:17:08Z","snapshot_observed_at":"2026-08-07T12:51:07.988679Z","submitted_at":"2025-06-16T16:17:08Z","title":"Ego-R1: Chain-of-Tool-Thought for Ultra-Long Egocentric Video Reasoning","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-07T00:34:36.092629Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2506.13654"},"observation_digest":"sha256:f81bf4871b2bb56aa0ccb6149452eb7c6c92f33da94fb7da5e5ebe2d8b2003e2","observation_id":"bdd91320-29dd-4a1d-9682-0486a7bc0404","resolution":{"observed_at":"2026-08-07T00:34:36.092629Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-06T17:26:40.342191Z","title":"Internvideo2.5: Empowering video mllms with long and rich context modeling,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10894","last_updated":"2025-07-15T01:20:22Z","snapshot_observed_at":"2026-08-11T01:18:30.551163Z","submitted_at":"2025-07-15T01:20:22Z","title":"NavComposer: Composing Language Instructions for Navigation Trajectories through Action-Scene-Object Modularization","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T17:26:40.342191Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2507.10894"},"observation_digest":"sha256:b065eaf621b43fbabc5e5bb2a809cd581be1fdec2857bbc1db0721d3a3d349ec","observation_id":"b26fa6d3-8e11-4c3e-9537-a920ae3bceae","resolution":{"observed_at":"2026-08-06T17:26:40.342191Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-06T14:36:29.583224Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.18531","last_updated":"2025-07-24T15:58:36Z","snapshot_observed_at":"2026-08-07T22:13:34.998544Z","submitted_at":"2025-07-24T15:58:36Z","title":"IntentVCNet: Bridging Spatio-Temporal Gaps for Intention-Oriented Controllable Video Captioning","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T14:36:29.583224Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2507.18531"},"observation_digest":"sha256:0c5225ad47ec1f00abed70a7781db00e6fbea3760acd473bb8d09d25a299f346","observation_id":"9e461aa4-bd38-4e8b-a4bb-e5313aa4cb95","resolution":{"observed_at":"2026-08-06T14:36:29.583224Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-06T05:15:43.265563Z","title":"Internvideo2","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.02094","last_updated":"2025-08-04T06:05:36Z","snapshot_observed_at":"2026-08-08T15:06:07.047501Z","submitted_at":"2025-08-04T06:05:36Z","title":"\"Harmless to You, Hurtful to Me!\": Investigating the Detection of Toxic Languages Grounded in the Perspective of Youth","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-06T05:15:43.265563Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2508.02094"},"observation_digest":"sha256:5c4c09a141280ab794b4f5472c2097f6206b679c700fd24154666073720a2ecc","observation_id":"537ae9d2-8b8b-4ee2-87fc-d2868ca37de7","resolution":{"observed_at":"2026-08-06T05:15:43.265563Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-06T05:12:38.444811Z","title":"Internvideo2","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.02095","last_updated":"2025-08-06T19:21:50Z","snapshot_observed_at":"2026-08-08T15:06:09.808299Z","submitted_at":"2025-08-04T06:06:06Z","title":"VLM4D: Towards Spatiotemporal Awareness in Vision Language Models","version":2},"reference_index":83,"source":"arxiv_source","source_observed_at":"2026-08-06T05:12:38.444811Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2508.02095"},"observation_digest":"sha256:aa4cb4401ede8c381d050afd5ae58f67cc51a7d6b1c87d4372cfd253841ff32d","observation_id":"0e018751-258d-4358-8242-05dba5a50bca","resolution":{"observed_at":"2026-08-06T05:12:38.444811Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-06T04:49:41.523576Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.03039","last_updated":"2025-08-05T03:33:24Z","snapshot_observed_at":"2026-08-07T14:34:47.687957Z","submitted_at":"2025-08-05T03:33:24Z","title":"VideoForest: Person-Anchored Hierarchical Reasoning for Cross-Video Question Answering","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T04:49:41.523576Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2508.03039"},"observation_digest":"sha256:53749a30057fedefa875ef50a5bc1b646530e1317eccaa5172c179d82726bccc","observation_id":"2ac3d27f-7c44-4e0a-be78-6d8ba40ee92e","resolution":{"observed_at":"2026-08-06T04:49:41.523576Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2509.15602","last_updated":"2026-04-16T06:31:32Z","snapshot_observed_at":"2026-08-03T02:26:40.337656Z","submitted_at":"2025-09-19T05:08:05Z","title":"TennisTV: Do Multimodal Large Language Models Understand Tennis Rallies?","version":5},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-18T16:40:16.630602Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2509.15602"},"observation_digest":"sha256:d4cb093fdddb3de313de3b5229973e4c840baa051182896eeb0dc5289093acd8","observation_id":"b5076d94-07c6-4475-8fb1-31568095960e","resolution":{"observed_at":"2026-05-18T16:41:37.834324Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2509.24943","last_updated":"2026-05-06T06:07:38Z","snapshot_observed_at":"2026-08-11T09:55:06.297723Z","submitted_at":"2025-09-29T15:42:55Z","title":"Perceive, Verify and Understand Long Video: Multi-Granular Perception and Active Verification via Interactive Agents","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-05-18T12:40:19.544260Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2509.24943"},"observation_digest":"sha256:311b425d5cf7ea04c470bf7378194cd886d57e92deeb0f7caffbba93669586a8","observation_id":"6c2e6875-f1a1-4088-bc1c-21e4d053fcd5","resolution":{"observed_at":"2026-05-18T12:41:22.584480Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-04T13:27:58.011777Z","title":"Internvideo2.5: Empowering video mllms with long and rich context modeling.arXiv preprint arXiv:2501.12386, 2025b","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2510.00705","last_updated":"2026-06-26T15:22:20Z","snapshot_observed_at":"2026-08-10T21:44:39.000633Z","submitted_at":"2025-10-01T09:20:51Z","title":"Training-free Uncertainty Guidance for Complex Visual Tasks with MLLMs","version":3},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-04T13:27:58.011777Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2510.00705"},"observation_digest":"sha256:9980364e2f69be3ab14f591cb46187036b054b58765f28534bf6758871410788","observation_id":"af50635d-a1d9-4977-9180-3439cbf037d9","resolution":{"observed_at":"2026-08-04T13:27:58.011777Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-04T07:23:07.235364Z","title":"Internvideo2","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2510.26113","last_updated":"2026-06-18T20:06:47Z","snapshot_observed_at":"2026-08-06T20:27:36.141566Z","submitted_at":"2025-10-30T03:53:22Z","title":"EgoExo-Con: Exploring View-Invariant Video Temporal Understanding","version":2},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-04T07:23:07.235364Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2510.26113"},"observation_digest":"sha256:e79bfb08f31062b6c817ad13d21022b2f6d348b26bd5e62c57d62e8b30e1adce","observation_id":"5b064333-15cc-4f7f-9284-4c40a6317838","resolution":{"observed_at":"2026-08-04T07:23:07.235364Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2512.03043","last_updated":"2026-04-28T12:07:36Z","snapshot_observed_at":"2026-08-10T21:08:10.223339Z","submitted_at":"2025-12-02T18:59:52Z","title":"OneThinker: All-in-one Reasoning Model for Image and Video","version":3},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-17T02:09:39.820651Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2512.03043"},"observation_digest":"sha256:3e37e514fc121fb4e515ec944cf93203b475568882d83bfb4906b701666485a8","observation_id":"125e64e3-7bd1-4b90-98ff-66ee8ca3c617","resolution":{"observed_at":"2026-05-17T02:52:20.835573Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2512.13511","last_updated":"2026-05-01T11:23:38Z","snapshot_observed_at":"2026-07-06T22:39:04.164003Z","submitted_at":"2025-12-15T16:38:59Z","title":"Adapting MLLMs for Nuanced Video Retrieval","version":3},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-05-16T22:20:09.051957Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2512.13511"},"observation_digest":"sha256:ef584aabb57b756f4d544200b83d3bec14707e1429a9478806a49455903f9016","observation_id":"f969c771-3294-4789-b06b-0369962a0a19","resolution":{"observed_at":"2026-05-17T02:52:20.835573Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-03T14:57:29.163851Z","title":"Internvideo2","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.18735","last_updated":"2026-05-25T12:55:14Z","snapshot_observed_at":"2026-08-07T05:11:54.627363Z","submitted_at":"2025-12-21T13:50:26Z","title":"$M^3-Verse$: A \"Spot the Difference\" Challenge for Large Multimodal Models","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-03T14:57:29.163851Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2512.18735"},"observation_digest":"sha256:bcf6ded02fd1b11e87ad2cd71c6362e1c8c63562817b797fd16724be610e898f","observation_id":"511079d9-318b-46a3-8e6e-da53b0d1ba2b","resolution":{"observed_at":"2026-08-03T14:57:29.163851Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2512.21334","last_updated":"2026-04-10T15:00:46Z","snapshot_observed_at":"2026-08-11T12:25:07.868269Z","submitted_at":"2025-12-24T18:59:36Z","title":"Streaming Video Instruction Tuning","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-16T19:44:11.032898Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2512.21334"},"observation_digest":"sha256:e48a104e866d14788e626301bd068d8a21a9439ee39090a4e635a8183f457b41","observation_id":"8eca61ed-2e08-4c71-9a6e-711a3e782706","resolution":{"observed_at":"2026-05-17T02:52:20.835573Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-07-13T21:37:55.887477Z","title":"Internvideo2","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.20190","last_updated":"2026-06-09T22:16:34Z","snapshot_observed_at":"2026-08-04T20:13:17.293244Z","submitted_at":"2026-03-20T17:59:25Z","title":"CoVR-R:Reason-Aware Composed Video Retrieval","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-07-13T21:37:55.887477Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2603.20190"},"observation_digest":"sha256:c27d175ee5854510bb680dd8175fb316c1dfe4015969d9fe58da2d54d8abbe0b","observation_id":"b5c2d4b6-449e-48e8-ba9b-73b57c33bf8e","resolution":{"observed_at":"2026-07-13T21:37:55.887477Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-07-13T15:29:31.834566Z","title":"arXiv preprint arXiv:2501.12386 (2025) 9, 10, 13, 2","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.29943","last_updated":"2026-07-10T15:58:11Z","snapshot_observed_at":"2026-08-07T09:23:25.859554Z","submitted_at":"2026-03-31T16:16:17Z","title":"Diagnosing Long-Video Quantitative Reasoning in Multimodal LLMs via Enumeration and Counting","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-07-13T15:29:31.834566Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2603.29943"},"observation_digest":"sha256:111bbcdb487c0bcef01a9f95b984bc3dec561c9dd4a1da7845b671529f402070","observation_id":"0eb35f2e-c345-452b-ac95-bdcc3ba2d825","resolution":{"observed_at":"2026-07-13T15:29:31.834566Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2604.04372","last_updated":"2026-04-06T02:43:03Z","snapshot_observed_at":"2026-07-06T22:53:20.122451Z","submitted_at":"2026-04-06T02:43:03Z","title":"Graph-to-Frame RAG: Visual-Space Knowledge Fusion for Training-Free and Auditable Video Reasoning","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-10T19:46:16.975267Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2604.04372"},"observation_digest":"sha256:15a2a53a78fc232a9784ec138d6743fec30cb46f5778e8f7fc415a46b7a8c0c1","observation_id":"3a8b1fb9-ff9f-42df-a6b5-7d0c193524d5","resolution":{"observed_at":"2026-05-17T02:52:20.835573Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2604.08337","last_updated":"2026-04-09T15:10:25Z","snapshot_observed_at":"2026-08-03T10:29:21.842229Z","submitted_at":"2026-04-09T15:10:25Z","title":"InstAP: Instance-Aware Vision-Language Pre-Train for Spatial-Temporal Understanding","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-10T18:06:33.139310Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2604.08337"},"observation_digest":"sha256:1d61367ca7402eba54fadf9607779fe28b84702cb3ea50dbc86ee17188a64967","observation_id":"9927fd3a-fbcf-4b98-8340-fe8b16e5cc32","resolution":{"observed_at":"2026-05-17T02:52:20.835573Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2604.08966","last_updated":"2026-04-10T05:10:15Z","snapshot_observed_at":"2026-07-06T22:57:59.555972Z","submitted_at":"2026-04-10T05:10:15Z","title":"How Should Video LLMs Output Time? An Analysis of Efficient Temporal Grounding Paradigms","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-10T16:50:48.559703Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2604.08966"},"observation_digest":"sha256:1816a287e6d139a696b2e1c901ca91c50953221959a5a8ef0726070db9daf933","observation_id":"d75ee4db-3bfd-4077-ada7-08f5479605c9","resolution":{"observed_at":"2026-05-17T02:52:20.835573Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2604.11240","last_updated":"2026-08-07T07:55:15Z","snapshot_observed_at":"2026-08-11T12:09:33.549424Z","submitted_at":"2026-04-13T09:44:52Z","title":"Decoupled Similarity for Task-Aware Token Pruning in Large Vision-Language Models","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-10T15:15:04.261855Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2604.11240"},"observation_digest":"sha256:9fc6432981351ac1e3416df592c4176e2825813aa4ad88009006e818c1753b2b","observation_id":"ba20b7fd-ddc6-422c-b8ae-4b189c29cccf","resolution":{"observed_at":"2026-05-17T02:52:20.835573Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2604.12335","last_updated":"2026-04-14T06:17:35Z","snapshot_observed_at":"2026-08-11T09:39:12.749021Z","submitted_at":"2026-04-14T06:17:35Z","title":"All in One: A Unified Synthetic Data Pipeline for Multimodal Video Understanding","version":1},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-05-10T15:26:55.369840Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2604.12335"},"observation_digest":"sha256:09c9e6bdf9988dbbdaadd9eb4073be35ea54cd6d24749f635ffac051b36c729b","observation_id":"bc29fe2b-741c-42aa-b69d-71c896e9466f","resolution":{"observed_at":"2026-05-17T02:52:20.835573Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2604.21873","last_updated":"2026-04-23T17:17:18Z","snapshot_observed_at":"2026-08-11T06:27:32.552043Z","submitted_at":"2026-04-23T17:17:18Z","title":"Grounding Video Reasoning in Physical Signals","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-09T22:29:21.240010Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2604.21873"},"observation_digest":"sha256:9e8cea1373914b62e8a6cdb1fbbd0bd7def8141eaf1a21b03f3a14db836177b7","observation_id":"bc44a8a6-a9d4-4297-a4aa-4a1f6216f892","resolution":{"observed_at":"2026-05-17T02:52:20.835573Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2605.00496","last_updated":"2026-05-01T08:12:26Z","snapshot_observed_at":"2026-08-03T21:34:47.954476Z","submitted_at":"2026-05-01T08:12:26Z","title":"High-Speed Vision Improves Zero-Shot Semantic Understanding of Human Actions","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-09T19:49:57.016399Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2605.00496"},"observation_digest":"sha256:c929a96214ca7aebc81bb93fd1bebbc9aa95e5600d836456a9e4f7d52cb592a0","observation_id":"fbce2764-2f30-44e6-9315-2647701224ee","resolution":{"observed_at":"2026-05-17T02:52:20.835573Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2605.04515","last_updated":"2026-05-06T05:48:56Z","snapshot_observed_at":"2026-08-06T20:21:26.793848Z","submitted_at":"2026-05-06T05:48:56Z","title":"From Priors to Perception: Grounding Video-LLMs in Physical Reality","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-08T17:41:23.233366Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2605.04515"},"observation_digest":"sha256:92e776091e7de2779df684fb29d1cbc048ccc68f54b3e7f962fab5cc42055b1c","observation_id":"c37adac5-757e-451d-8091-6c3842fc4730","resolution":{"observed_at":"2026-05-17T02:52:20.835573Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2605.06094","last_updated":"2026-05-22T03:20:58Z","snapshot_observed_at":"2026-08-11T01:16:46.714598Z","submitted_at":"2026-05-07T12:13:15Z","title":"VISD: Enhancing Video Reasoning via Structured Self-Distillation","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-08T14:06:27.953376Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2605.06094"},"observation_digest":"sha256:9811d92096c57993e52d96b285275abc774fd4c5c28555c6b17f395e2dd8ec67","observation_id":"923b11be-4673-4dcc-ab52-11d504621cea","resolution":{"observed_at":"2026-05-17T02:52:20.835573Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2605.06094","last_updated":"2026-05-22T03:20:58Z","snapshot_observed_at":"2026-08-11T01:16:46.714598Z","submitted_at":"2026-05-07T12:13:15Z","title":"VISD: Enhancing Video Reasoning via Structured Self-Distillation","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-11T01:49:41.654207Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2605.06094"},"observation_digest":"sha256:27e9105c300abccf7dc7828e9b490186fc102d68d0644b6de6f62ef4c0c2f83d","observation_id":"cf8b4a35-bc2a-4db0-b47f-2e8b5783d3fa","resolution":{"observed_at":"2026-05-17T02:52:20.835573Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2605.06094","last_updated":"2026-05-22T03:20:58Z","snapshot_observed_at":"2026-08-11T01:16:46.714598Z","submitted_at":"2026-05-07T12:13:15Z","title":"VISD: Enhancing Video Reasoning via Structured Self-Distillation","version":3},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-12T03:35:59.553683Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2605.06094"},"observation_digest":"sha256:70fc12357803a62cc0f831356b4c4558ad716ec94b17ba88bd6d93807f613a82","observation_id":"212bc8f4-2e67-4968-b456-18f3da2e09b4","resolution":{"observed_at":"2026-05-17T02:52:20.835573Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2605.06094","last_updated":"2026-05-22T03:20:58Z","snapshot_observed_at":"2026-08-11T01:16:46.714598Z","submitted_at":"2026-05-07T12:13:15Z","title":"VISD: Enhancing Video Reasoning via Structured Self-Distillation","version":4},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-25T06:08:19.956833Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2605.06094"},"observation_digest":"sha256:0843dfff18e53170a5df647f0616bd53b96be80a2b75b4ebea56152e44e42a60","observation_id":"35b43260-bc1a-4853-8630-7f139cb4eca7","resolution":{"observed_at":"2026-05-25T06:10:23.843719Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2605.07897","last_updated":"2026-05-08T15:40:40Z","snapshot_observed_at":"2026-08-09T07:41:26.425560Z","submitted_at":"2026-05-08T15:40:40Z","title":"Semantic-Aware Adaptive Visual Memory for Streaming Video Understanding","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-11T02:12:20.651111Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2605.07897"},"observation_digest":"sha256:2e1cfe96ad9bdedc46f2bc6e88230299800ec9a12b4df403caf44f0465f4738a","observation_id":"f258556c-d103-4168-afd5-85ef3680fb2f","resolution":{"observed_at":"2026-05-17T02:52:20.835573Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2605.09874","last_updated":"2026-05-11T01:59:59Z","snapshot_observed_at":"2026-07-30T14:13:44.494421Z","submitted_at":"2026-05-11T01:59:59Z","title":"EgoMemReason: A Memory-Driven Reasoning Benchmark for Long-Horizon Egocentric Video Understanding","version":1},"reference_index":105,"source":"arxiv_source","source_observed_at":"2026-05-12T04:37:46.090777Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2605.09874"},"observation_digest":"sha256:66bc96cee8fc094625bd951036c2b82549bc73dfd0b3443e50d6d365c1a90ccf","observation_id":"b04166d4-610d-4ffc-9b28-986a6891da3d","resolution":{"observed_at":"2026-05-17T02:52:20.835573Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2605.09904","last_updated":"2026-05-12T03:09:23Z","snapshot_observed_at":"2026-07-06T23:21:54.140120Z","submitted_at":"2026-05-11T02:47:59Z","title":"TOC-Bench: A Temporal Object Consistency Benchmark for Video Large Language Models","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-12T04:13:21.487431Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2605.09904"},"observation_digest":"sha256:3ee2805ccbc227be851d4ffb3ec56aa6e59ebffb0955856a79243b51d24971f7","observation_id":"b81a58f8-962f-4907-9aa2-117c800457fc","resolution":{"observed_at":"2026-05-17T02:52:20.835573Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2605.09904","last_updated":"2026-05-12T03:09:23Z","snapshot_observed_at":"2026-07-06T23:21:54.140120Z","submitted_at":"2026-05-11T02:47:59Z","title":"TOC-Bench: A Temporal Object Consistency Benchmark for Video Large Language Models","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-13T06:53:42.726350Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2605.09904"},"observation_digest":"sha256:4834b912653f5b0719c8e12064ebf3641617e4607c33464796dadd7d064637b4","observation_id":"0c3eca2b-ff6e-4ab1-810b-149a51feefe4","resolution":{"observed_at":"2026-05-17T02:52:20.835573Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2605.14310","last_updated":"2026-05-14T03:22:30Z","snapshot_observed_at":"2026-08-11T06:03:31.469444Z","submitted_at":"2026-05-14T03:22:30Z","title":"CoRDS: Coreset-based Representative and Diverse Selection for Streaming Video Understanding","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-15T02:31:19.662372Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2605.14310"},"observation_digest":"sha256:9bd8823908e550c96154b35d0a6c1d937ea9527aa9d5f66116ddb5e827a90d48","observation_id":"ede6c788-8b9f-49a7-8cbd-e05aa70ac9c4","resolution":{"observed_at":"2026-05-17T02:52:20.835573Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2605.14733","last_updated":"2026-05-14T11:56:14Z","snapshot_observed_at":"2026-08-06T01:04:24.065010Z","submitted_at":"2026-05-14T11:56:14Z","title":"Video-Zero: Self-Evolution Video Understanding","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-06-30T21:32:16.939563Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2605.14733"},"observation_digest":"sha256:fff036cffa60e604ac791e8711524a2083db00d2465f9dbc1fcb2dd01c44397b","observation_id":"58ae6413-6598-4b95-937b-02f8a737aa6d","resolution":{"observed_at":"2026-06-30T21:35:04.405551Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2605.17260","last_updated":"2026-05-23T08:41:12Z","snapshot_observed_at":"2026-07-06T23:28:16.927009Z","submitted_at":"2026-05-17T05:02:52Z","title":"LiteFrame: Efficient Vision Encoders Unlock Frame Scaling in Video LLMs","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-20T14:54:12.264474Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2605.17260"},"observation_digest":"sha256:958cc9251688cdfb44ba79822678692c04dfbebd6e17fe8a1bb5df15d8b4e43b","observation_id":"fef7ad78-f74b-42a8-9b45-2ff7f068697c","resolution":{"observed_at":"2026-05-20T14:58:24.740744Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2605.17260","last_updated":"2026-05-23T08:41:12Z","snapshot_observed_at":"2026-07-06T23:28:16.927009Z","submitted_at":"2026-05-17T05:02:52Z","title":"LiteFrame: Efficient Vision Encoders Unlock Frame Scaling in Video LLMs","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-30T19:45:01.311800Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2605.17260"},"observation_digest":"sha256:2864110f870312bf34472204865bfd739049d8b16abc35287f76e09b9cc6cc3d","observation_id":"83ae1637-3a12-4761-a6c8-4f0e0de9c6c0","resolution":{"observed_at":"2026-07-01T14:55:47.286739Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2605.17283","last_updated":"2026-05-17T06:39:05Z","snapshot_observed_at":"2026-08-08T15:14:20.631028Z","submitted_at":"2026-05-17T06:39:05Z","title":"OProver: A Unified Framework for Agentic Formal Theorem Proving","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-05-20T14:43:46.517807Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2605.17283"},"observation_digest":"sha256:3c85b3de92bd126b8c973b6d6b5440dd3fc35f4301b513eb8d7deff56cd9630f","observation_id":"3face163-f13a-4359-8295-62fd8ad61842","resolution":{"observed_at":"2026-05-20T14:48:23.444378Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2605.22269","last_updated":"2026-05-21T10:13:03Z","snapshot_observed_at":"2026-08-02T11:53:00.337572Z","submitted_at":"2026-05-21T10:13:03Z","title":"MuKV: Multi-Grained KV Cache Compression for Long Streaming Video Question-Answering","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-22T07:16:33.817183Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2605.22269"},"observation_digest":"sha256:a68b7ccfeb02b50e234894ab099888d0f46a1ada412f5f994ebb2ce2050c6bf2","observation_id":"bf36e788-09c0-43e7-8528-2b51fd61e1ed","resolution":{"observed_at":"2026-05-22T07:21:13.185135Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2605.27074","last_updated":"2026-05-26T14:23:25Z","snapshot_observed_at":"2026-08-02T20:44:04.329111Z","submitted_at":"2026-05-26T14:23:25Z","title":"IPIBench: Evaluating Interactive Proactive Intelligence of MLLMs under Continuous Streams","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-06-29T18:24:57.881644Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2605.27074"},"observation_digest":"sha256:4efd757c6c2288b51e7632ad98e9fcda7f02b384501a9299fc3ca9984db70c9a","observation_id":"eb83c77e-b09d-4845-80da-8151f417c158","resolution":{"observed_at":"2026-06-29T18:33:51.057493Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2605.28229","last_updated":"2026-05-27T09:43:06Z","snapshot_observed_at":"2026-08-05T02:10:57.115222Z","submitted_at":"2026-05-27T09:43:06Z","title":"VidPrism: Heterogeneous Mixture of Experts for Image-to-Video Transfer","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-06-29T13:01:42.880738Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2605.28229"},"observation_digest":"sha256:16731056ef459b35c65fcb05c4d42bf1f7d5c37e15e8a003f5381f572a6c88ca","observation_id":"bb48f86c-314b-441a-9a64-3c18019f3738","resolution":{"observed_at":"2026-06-29T13:03:25.898452Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2605.29858","last_updated":"2026-05-28T12:39:04Z","snapshot_observed_at":"2026-08-01T18:17:02.617293Z","submitted_at":"2026-05-28T12:39:04Z","title":"Masked Diffusion Vision-Language Models for Temporal Action Localization","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-29T07:56:52.524815Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2605.29858"},"observation_digest":"sha256:8ee3ea44148ab38bc1e69a98c5a4e1f5976479a0ed26ec1a4ccd6e15200066f0","observation_id":"670c8e33-42ac-4af3-9923-6d3663bc5fc1","resolution":{"observed_at":"2026-06-29T08:03:13.693211Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2606.05718","last_updated":"2026-06-04T05:18:13Z","snapshot_observed_at":"2026-08-02T23:26:06.327596Z","submitted_at":"2026-06-04T05:18:13Z","title":"ViCuR: Visual Cues as Recoverable Privilege for Multimodal On-Policy Distillation","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-28T01:59:15.154873Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2606.05718"},"observation_digest":"sha256:41f85ad63c14d0885237dee13eeae529f455fa02866411bb567cb7477cfceb39","observation_id":"6e7de1fb-0a2f-4b9c-a6a3-039326139e56","resolution":{"observed_at":"2026-07-02T12:36:56.986840Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2606.06532","last_updated":"2026-06-03T17:47:49Z","snapshot_observed_at":"2026-08-07T12:55:44.925060Z","submitted_at":"2026-06-03T17:47:49Z","title":"GOPAgen: Motion-Aware and Efficient Agentic Long-Video Understanding with Structural Memory and Hierarchical Reasoning","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-06-28T06:33:32.090913Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2606.06532"},"observation_digest":"sha256:21f10be42f0b7b163b6feb5671f3c6f514f9be5a1c80f9e240f2fc3d160e85b3","observation_id":"0f97e9b7-e167-413e-886c-b5181e0c233a","resolution":{"observed_at":"2026-07-02T07:56:47.318780Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-08-10T11:26:35.969932Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:2487e564d6c2c8b0905e09cd5cb575eb133d9a257fb4a788c7e5adebbf27ad03","observation_id":"b62462ba-6f47-43e3-948c-df820a424819","resolution":{"observed_at":"2026-07-02T17:07:12.785043Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2606.10401","last_updated":"2026-06-10T09:54:52Z","snapshot_observed_at":"2026-07-06T23:49:37.345711Z","submitted_at":"2026-06-09T04:20:08Z","title":"CoCoSI: Collaborative Cognitive Map Construction for Spatial Intelligence","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-27T14:17:08.713863Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2606.10401"},"observation_digest":"sha256:917ed91904ed2267d169614df836ad7dd2aeeb610c2aaf43a419a74a9f82a405","observation_id":"76fafa33-ac4f-443b-845c-5b6137108df7","resolution":{"observed_at":"2026-07-03T03:57:38.936566Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2606.11792","last_updated":"2026-06-11T05:33:10Z","snapshot_observed_at":"2026-08-05T06:18:17.353570Z","submitted_at":"2026-06-10T08:25:09Z","title":"MultiToP: Learning to Patch Visual Tokens to Mitigate Hallucinations in Video Large Multimodal Models","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-06-27T10:02:58.341050Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2606.11792"},"observation_digest":"sha256:a5675ea008486faeb387875df2494f4c99bbb28879f4b4c6bcdffebea8587f60","observation_id":"8ae19305-c3cc-420b-abc8-6aa4de49f69c","resolution":{"observed_at":"2026-07-03T10:27:56.120244Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2606.12195","last_updated":"2026-06-10T15:17:08Z","snapshot_observed_at":"2026-08-01T02:09:41.655807Z","submitted_at":"2026-06-10T15:17:08Z","title":"InternVideo3: Agentify Foundation Models with Multimodal Contextual Reasoning","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-06-27T09:48:27.652901Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2606.12195"},"observation_digest":"sha256:d324eb245359d960589a082dc54494001d4a5c57412308efae1d771966022732","observation_id":"067fdded-660c-4c46-95ad-f2f2b80fb3ae","resolution":{"observed_at":"2026-07-03T10:48:03.182059Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-08-08T11:16:26.006355Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:119e20c9bc9c525f4b5c86d62a06f78eed762cf4bcb79c5730f4ffd1e2fb7500","observation_id":"40caaa14-d8ec-4074-a426-ddab0dadf133","resolution":{"observed_at":"2026-07-03T20:38:56.157736Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2606.27988","last_updated":"2026-06-26T11:35:01Z","snapshot_observed_at":"2026-07-07T00:02:09.217954Z","submitted_at":"2026-06-26T11:35:01Z","title":"Latent Visual Diffusion Reasoning with Monte Carlo Tree Search","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-29T04:55:37.555809Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2606.27988"},"observation_digest":"sha256:cd9d96fe2c45763abd9f9e652a344cff862bfcbadf9e98a13e1dc7f04bbc5fde","observation_id":"c15dff4b-100b-4359-9947-c668c84245f7","resolution":{"observed_at":"2026-06-29T19:03:52.403983Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2606.29531","last_updated":"2026-06-28T17:54:55Z","snapshot_observed_at":"2026-08-02T23:05:17.329812Z","submitted_at":"2026-06-28T17:54:55Z","title":"MotionAtlas: Detailed Region Captioning for Motion-Centric Videos","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-30T07:21:46.783970Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2606.29531"},"observation_digest":"sha256:4aa8b7bc48b3cead74a60f45581af063fd8e591ad11064de183d7950c614162a","observation_id":"74cbdf07-f726-49e6-bbac-7ad254b47189","resolution":{"observed_at":"2026-06-30T07:24:21.122131Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2606.31187","last_updated":"2026-06-30T06:16:17Z","snapshot_observed_at":"2026-07-07T00:04:58.486133Z","submitted_at":"2026-06-30T06:16:17Z","title":"Learning to Deny: Action Denial in Multimodal Large Language Models","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-07-01T06:21:09.996386Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2606.31187"},"observation_digest":"sha256:843cc9135330ae09c7136d7b35d4a3bfc8eafef500c58ddd933e2a01d058a56d","observation_id":"739ba1f6-1d06-49be-a72b-87a3066511ab","resolution":{"observed_at":"2026-07-01T09:45:39.577296Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2606.31326","last_updated":"2026-06-30T08:29:29Z","snapshot_observed_at":"2026-07-07T00:05:02.977105Z","submitted_at":"2026-06-30T08:29:29Z","title":"Bridging Video Understanding and Generation in a Unified Framework","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-07-01T05:57:54.653504Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2606.31326"},"observation_digest":"sha256:e86ddfd6775607f39033329fabb708eda840b0ebf47454cbe638304f011b8066","observation_id":"f29c3843-0f5c-4a33-83c4-b1b4a5e870e8","resolution":{"observed_at":"2026-07-01T10:05:40.298995Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":"2501.12386","doi":"10.48550/arxiv.2501.12386","metadata_source":"pith","pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","venue":"cs.CV","work_id":"5801b83f-d5f4-4020-bad2-4bc47f71fe7d","year":2025},"citing_paper":{"arxiv_id":"2607.02096","last_updated":"2026-07-02T12:32:53Z","snapshot_observed_at":"2026-07-07T00:07:33.184524Z","submitted_at":"2026-07-02T12:32:53Z","title":"LongEgoRefer: A Benchmark for Long-Form Egocentric Video Referring Expression Comprehension","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-07-03T15:41:03.548799Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2607.02096"},"observation_digest":"sha256:4a983460d0cd917bef48655d55bcec84e3f44b655213408b7db0639711252835","observation_id":"5add77d4-1b8a-497d-908a-e85aa98bd5ce","resolution":{"observed_at":"2026-07-03T15:48:35.012694Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-07-12T05:16:19.750976Z","title":"5: Empowering video mllms with long and rich context modeling","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.03043","last_updated":"2026-07-03T07:33:28Z","snapshot_observed_at":"2026-08-07T12:44:29.845003Z","submitted_at":"2026-07-03T07:33:28Z","title":"Natural Language Camera Movement Understanding","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-07-12T05:16:19.750976Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2607.03043"},"observation_digest":"sha256:5f85255dfb331b7014edde361b2e00f44fee98bd4f09ae9ae21a87eb4f1c2850","observation_id":"5e8d3556-65e6-4e26-8f18-241baa4085a2","resolution":{"observed_at":"2026-07-12T05:16:19.750976Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-07-12T01:03:30.708082Z","title":"Internvideo2.5: Empowering video mllms with long and rich context modeling.arXiv preprint arXiv:2501.12386, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.03633","last_updated":"2026-07-03T23:17:58Z","snapshot_observed_at":"2026-08-06T13:13:24.284371Z","submitted_at":"2026-07-03T23:17:58Z","title":"Probing Identity-Specific Motion Signatures: A Controlled Diagnostic Study","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-12T01:03:30.708082Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2607.03633"},"observation_digest":"sha256:2bcedd922a30bd0b16cf8d5f3a9a643224fda239cac86f32509be2acf8c7cba6","observation_id":"e882bf37-90d4-499d-b6eb-528ea4968e99","resolution":{"observed_at":"2026-07-12T01:03:30.708082Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-02T06:34:30.007079Z","title":"Internvideo2.5: Empower- 11 ing video mllms with long and rich context modeling.arXiv preprint arXiv:2501.12386, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.12503","last_updated":"2026-07-15T03:49:06Z","snapshot_observed_at":"2026-08-06T20:14:51.309947Z","submitted_at":"2026-07-14T08:35:07Z","title":"DynTrace: Tracking Dynamic Object Evidence for 4D Spatio-Temporal Reasoning in MLLMs","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-02T06:34:30.007079Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2607.12503"},"observation_digest":"sha256:2d1b4aa75cb14d445010cb932235fc9b2c3b7c3b88565c062ac6c2a4e3315c13","observation_id":"d0b6ef65-6f2c-4d51-b3ef-f0f043eb7e6f","resolution":{"observed_at":"2026-08-02T06:34:30.007079Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-02T00:44:40.385436Z","title":"Internvideo2.5: Empowering video mllms with long and rich context modeling","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14935","last_updated":"2026-07-16T12:47:59Z","snapshot_observed_at":"2026-08-09T10:08:14.840784Z","submitted_at":"2026-07-16T12:47:59Z","title":"VideoChat3: Fully Open Video MLLM for Efficient and Generalist Video Understanding","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-02T00:44:40.385436Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2607.14935"},"observation_digest":"sha256:d60bd7dfd11e04c24ff8b9dc41c34cab1b0f461365a0b0d972e4817ab1c5933e","observation_id":"dbcfc400-3eab-4288-add5-b74a12a64c15","resolution":{"observed_at":"2026-08-02T00:44:40.385436Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-02T07:40:00.692260Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16284","last_updated":"2026-08-06T06:01:52Z","snapshot_observed_at":"2026-08-09T23:09:55.314603Z","submitted_at":"2026-07-10T12:31:27Z","title":"MAC 2026: Advancing Micro-Action Analysis Towards Fine-Grained Understanding","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-02T07:40:00.692260Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2607.16284"},"observation_digest":"sha256:40207e39b58f770a9f1ec936bbf9d51f7734bf64d2133c7088ded90b6435afac","observation_id":"aa7e5358-e336-43c6-930b-f843dab8d307","resolution":{"observed_at":"2026-08-02T07:40:00.692260Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-01T14:37:47.442834Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.18716","last_updated":"2026-07-21T05:16:36Z","snapshot_observed_at":"2026-08-05T20:25:01.690963Z","submitted_at":"2026-07-21T05:16:36Z","title":"Continual Video-MLLM Adaptation over Evolving Domains","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-01T14:37:47.442834Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2607.18716"},"observation_digest":"sha256:7b2577eddfaf422a2d905e448e618c8e4a0a2780b6a84fb7a6c8cef1ab3041a9","observation_id":"a282df78-dfdb-4c02-a1fa-68bff1bba73f","resolution":{"observed_at":"2026-08-01T14:37:47.442834Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-01T08:23:24.495566Z","title":"Internvideo2. 5: Empowering video mllms with long and rich context modeling,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.21151","last_updated":"2026-07-26T23:23:12Z","snapshot_observed_at":"2026-08-06T16:53:45.184525Z","submitted_at":"2026-07-23T10:35:38Z","title":"V-DEAL: Diagnosing Video Safety De-Calibration as an Understanding-Refusal Coupling Failure","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-01T08:23:24.495566Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2607.21151"},"observation_digest":"sha256:b8090a8dcfcb49f4238c1a8fcfe84d53ffd01a15c1d751c40e236e210e6a6cd1","observation_id":"ff3267ed-70a9-484c-bfc9-b1e7b5a9304d","resolution":{"observed_at":"2026-08-01T08:23:24.495566Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-07-31T23:32:54.425149Z","title":"Internvideo2.5: Empowering video mllms with long and rich context modeling.arXiv preprint arXiv:2501.12386, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.23951","last_updated":"2026-07-27T02:56:43Z","snapshot_observed_at":"2026-07-31T23:32:36.040113Z","submitted_at":"2026-07-27T02:56:43Z","title":"TimePLE: Rethinking Temporal Representation for Video Temporal Grounding","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-07-31T23:32:54.425149Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2607.23951"},"observation_digest":"sha256:1e92673047ec3a973d07657d9776c65dbf3f56caa80c4b977fa0eaa382e8530a","observation_id":"2f64894f-968d-4087-8589-5ad2eec00747","resolution":{"observed_at":"2026-07-31T23:32:54.425149Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-01T02:59:25.397951Z","title":"arXiv preprint arXiv:2501.12386 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.25266","last_updated":"2026-07-30T21:59:18Z","snapshot_observed_at":"2026-08-05T23:11:09.274885Z","submitted_at":"2026-07-28T04:09:49Z","title":"FORGE: Frame Orthogonality in Relevance Geometry for Long-Form Video Understanding","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-01T02:59:25.397951Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2607.25266"},"observation_digest":"sha256:1237ea2ac1148d4069b0bccc0026105bc8a6cc9b8c83c179e6cf5f8970e5be92","observation_id":"f2e2663e-7d84-4879-84ce-54ed7852962f","resolution":{"observed_at":"2026-08-01T02:59:25.397951Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-03T01:46:11.788344Z","title":"arXiv preprint arXiv:2501.12386 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.25266","last_updated":"2026-07-30T21:59:18Z","snapshot_observed_at":"2026-08-05T23:11:09.274885Z","submitted_at":"2026-07-28T04:09:49Z","title":"FORGE: Frame Orthogonality in Relevance Geometry for Long-Form Video Understanding","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-03T01:46:11.788344Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2607.25266"},"observation_digest":"sha256:ad1b41ed82564ede9a5f40bb1c8edc0f3282466c608ca4cb3d4d4d489af9af9b","observation_id":"f0c17133-5f7a-4704-b073-b60832060a27","resolution":{"observed_at":"2026-08-03T01:46:11.788344Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-07-31T06:37:32.975156Z","title":"InternVideo2.5: Empowering video MLLMs with long and rich context modeling,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.28463","last_updated":"2026-07-30T16:23:10Z","snapshot_observed_at":"2026-08-06T10:12:19.900175Z","submitted_at":"2026-07-30T16:23:10Z","title":"VisualRouter: Query-Grounded Visual Sampling for Long Video Understanding","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-07-31T06:37:32.975156Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2607.28463"},"observation_digest":"sha256:8b9f18b5ff9e812b7c67ad08bc4649471b058ac08e3425942d45b79938ca6fc1","observation_id":"32732981-9dac-4e5e-b682-39fa7b19ab3a","resolution":{"observed_at":"2026-07-31T06:37:32.975156Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-06T00:31:40.530514Z","title":"InProceedings of the 3rd ACM SIGPLAN InternationalWorkshoponMachineLearningandProgram- ming Languages (MAPL), 10–19","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.01169","last_updated":"2026-08-02T11:44:51Z","snapshot_observed_at":"2026-08-11T01:21:42.500512Z","submitted_at":"2026-08-02T11:44:51Z","title":"Think in Sets for Streaming Video Token Compression","version":1},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-06T00:31:40.530514Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2608.01169"},"observation_digest":"sha256:188be750be97ccb01fcfa26cf8149850a1bce9afcd3857e5817ae129675fafec","observation_id":"ac560be2-623d-42f1-83e5-80d01c5921d6","resolution":{"observed_at":"2026-08-06T00:31:40.530514Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2501.12386/citation-record","integrity":"/paper/2501.12386/integrity","json":"/paper/2501.12386/citation-record.json","paper":"/paper/2501.12386"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.03575","last_updated":"2025-07-09T19:35:31Z","snapshot_observed_at":"2026-08-03T00:21:10.886100Z","submitted_at":"2025-01-07T06:55:50Z","title":"Cosmos World Foundation Model Platform for Physical AI","version":3},"cited_work":{"arxiv_id":"2501.03575","doi":"10.48550/arxiv.2501.03575","metadata_source":"pith","pith_arxiv_id":"2501.03575","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Cosmos World Foundation Model Platform for Physical AI","venue":"cs.CV","work_id":"a2dba24c-318d-476a-8b21-4289c265810c","year":2025},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2501.03575","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:efbecdcdfeba46af9cf89df4ce21b109993e83ec4fc40c990c1068bb929f929e","observation_id":"07d9e05a-eb3a-484f-b201-1502b89bf29a","resolution":{"observed_at":"2026-05-17T02:52:20.670876Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-07-11T15:53:29.560535+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T15:53:29.560535+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16609","last_updated":"2023-09-28T17:07:49Z","snapshot_observed_at":"2026-08-09T21:25:20.369782Z","submitted_at":"2023-09-28T17:07:49Z","title":"Qwen Technical Report","version":1},"cited_work":{"arxiv_id":"2309.16609","doi":"10.48550/arxiv.2309.16609","metadata_source":"pith","pith_arxiv_id":"2309.16609","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen Technical Report","venue":"cs.CL","work_id":"bb1fd52f-6b2f-437c-9516-37bdf6eb9be8","year":2023},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2309.16609","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:efcc0564701fa0091df405938c7557620613152f8651e87b5d6d75ae27cb139c","observation_id":"4e2e81ef-2ec9-40a3-be2b-9b2916aabacf","resolution":{"observed_at":"2026-05-17T02:52:20.675691Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-07-15T23:50:15.620681+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-15T23:50:15.620681+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.19603","last_updated":"2024-09-29T07:47:15Z","snapshot_observed_at":"2026-08-10T23:14:35.987685Z","submitted_at":"2024-09-29T07:47:15Z","title":"One Token to Seg Them All: Language Instructed Reasoning Segmentation in Videos","version":1},"cited_work":{"arxiv_id":"2409.19603","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.19603","snapshot_observed_at":"2026-07-03T10:48:03.088092Z","title":"One token to seg them all: Language instructed reasoning segmentation in videos","venue":null,"work_id":"3968ae62-2a48-4da8-8638-9234980a83dd","year":null},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2409.19603","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:c15dccae5b3f916f70f44af375757ccdb30bda55b27c5ebfb09f1bc0316d4644","observation_id":"939ad889-be6d-4612-ac51-676f8947dfee","resolution":{"observed_at":"2026-05-17T02:52:20.680766Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.09461","last_updated":"2023-03-01T19:45:11Z","snapshot_observed_at":"2026-07-06T14:06:56.291161Z","submitted_at":"2022-10-17T22:23:40Z","title":"Token Merging: Your ViT But Faster","version":3},"cited_work":{"arxiv_id":"2210.09461","doi":"10.48550/arxiv.2210.09461","metadata_source":"pith","pith_arxiv_id":"2210.09461","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Token Merging: Your ViT But Faster","venue":"cs.CV","work_id":"528509bc-2611-4e7f-a772-ea14d25b6dae","year":2022},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2210.09461","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:8f42f5aa24df0f1e0006f33e555523da8f14d76c3110daff61aaa6dd9741bee1","observation_id":"879cafa7-3662-4c5c-a00f-474d3eb50d28","resolution":{"observed_at":"2026-05-17T02:52:20.685487Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17297","last_updated":"2024-03-26T00:53:24Z","snapshot_observed_at":"2026-08-02T11:10:24.263044Z","submitted_at":"2024-03-26T00:53:24Z","title":"InternLM2 Technical Report","version":1},"cited_work":{"arxiv_id":"2403.17297","doi":"10.48550/arxiv.2403.17297","metadata_source":"pith","pith_arxiv_id":"2403.17297","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternLM2 Technical Report","venue":"cs.CL","work_id":"dfa13e0e-1c3c-4fb6-943d-a19945bacdbe","year":2024},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2403.17297","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:46487386a15bc036b1f5bd5ae9e21517d93e6777cfb7f3ae1d181306776db80a","observation_id":"1e93dfae-f8d2-48cb-bd97-4d65717864ef","resolution":{"observed_at":"2026-05-17T02:52:20.689675Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.04998","last_updated":"2024-11-07T18:59:16Z","snapshot_observed_at":"2026-07-06T19:46:54.707852Z","submitted_at":"2024-11-07T18:59:16Z","title":"HourVideo: 1-Hour Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2411.04998","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.04998","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"HourVideo: 1-hour video-language understanding","venue":null,"work_id":"8a5f3847-96f0-4e8c-aa50-0efe6ccf9dab","year":2024},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2411.04998","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:0616fd9e5129aa1e3949e741239e66c8392712a1688d1722620ceda1cc73a23a","observation_id":"c04387ad-e5a2-4f71-b3a1-cb412a049b00","resolution":{"observed_at":"2026-05-17T02:52:20.694510Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.11684","last_updated":"2024-06-17T07:55:59Z","snapshot_observed_at":"2026-07-06T17:31:53.996417Z","submitted_at":"2024-02-18T19:26:49Z","title":"ALLaVA: Harnessing GPT4V-Synthesized Data for Lite Vision-Language Models","version":2},"cited_work":{"arxiv_id":"2402.11684","doi":null,"metadata_source":"pith","pith_arxiv_id":"2402.11684","snapshot_observed_at":"2026-07-10T17:07:25.737454Z","title":"ALLaVA: Harnessing GPT4V-Synthesized Data for Lite Vision-Language Models","venue":"cs.CL","work_id":"4cf5f4e3-c59a-4ccb-a655-563157a9ce74","year":2024},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2402.11684","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:83f20761796c715a45cb2ee275a431253e5b8a66eaba3c6a2e89e0a0a342404f","observation_id":"be7b2f50-db56-46cf-a2b5-3f3ee2c0b42b","resolution":{"observed_at":"2026-05-23T22:20:22.000193Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.07719","last_updated":"2024-07-02T09:03:26Z","snapshot_observed_at":"2026-07-06T18:13:29.693360Z","submitted_at":"2024-05-13T13:08:02Z","title":"USP: A Unified Sequence Parallelism Approach for Long Context Generative AI","version":5},"cited_work":{"arxiv_id":"2405.07719","doi":"10.48550/arxiv.2405.07719","metadata_source":"pith","pith_arxiv_id":"2405.07719","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A unified sequence parallelism approach for long context generative ai","venue":"cs.LG","work_id":"277d0fdf-d9cc-4497-ad08-04aba6226c36","year":2024},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2405.07719","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:3cba234c846aa0f719dd16d239c9461d77390efa5a780784a2b8cf23cd72de77","observation_id":"5e544394-3655-4db0-a292-74efbb8a418d","resolution":{"observed_at":"2026-05-17T02:52:20.705108Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14515","last_updated":"2024-10-30T13:38:10Z","snapshot_observed_at":"2026-08-10T10:57:30.502675Z","submitted_at":"2024-06-20T17:26:01Z","title":"MMBench-Video: A Long-Form Multi-Shot Benchmark for Holistic Video Understanding","version":3},"cited_work":{"arxiv_id":"2406.14515","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.14515","snapshot_observed_at":"2026-06-28T23:12:46.636950Z","title":"Mmbench-video: A long-form multi-shot benchmark for holistic video under- standing","venue":null,"work_id":"0108342f-f183-4a3e-930e-337bd83a6601","year":2024},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2406.14515","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:8716f191366900dc85bfdad8e049ae0b39637b07627715e146597b851fc546ff","observation_id":"bcd3f53a-9953-4561-8386-c85bc699baec","resolution":{"observed_at":"2026-05-17T02:52:20.710199Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-10T07:39:00.084462Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":"2408.14023","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-07-04T09:39:46.765108Z","title":"Video-ccam: Enhancing video-language understanding with causal cross-attention masks for short and long videos","venue":null,"work_id":"2ed72cbe-f6a7-4149-911f-be5ff9cb6d0e","year":2024},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:87c4ecc36e07cef04c70acc15da798dbbffa825d530128f7601430ae15e355f9","observation_id":"a1e3ce7b-6cf8-4833-9414-91be823de78e","resolution":{"observed_at":"2026-05-17T02:52:20.714773Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12793","last_updated":"2024-07-30T03:58:11Z","snapshot_observed_at":"2026-08-07T13:56:34.167869Z","submitted_at":"2024-06-18T16:58:21Z","title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","version":2},"cited_work":{"arxiv_id":"2406.12793","doi":"10.48550/arxiv.2406.12793","metadata_source":"pith","pith_arxiv_id":"2406.12793","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","venue":"cs.CL","work_id":"de9ce5af-0d8d-4b94-9793-64968d9bc06d","year":2024},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2406.12793","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:2615aaaba954e610c8db0a3b5f35fc0e3360ea541d649347c1db6b08752f3466","observation_id":"40ea47f5-5c90-42a4-879d-4c6e346af6a6","resolution":{"observed_at":"2026-05-17T02:52:20.718503Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.14509","last_updated":"2023-10-04T16:51:13Z","snapshot_observed_at":"2026-08-04T19:27:31.715261Z","submitted_at":"2023-09-25T20:15:57Z","title":"DeepSpeed Ulysses: System Optimizations for Enabling Training of Extreme Long Sequence Transformer Models","version":2},"cited_work":{"arxiv_id":"2309.14509","doi":"10.48550/arxiv.2309.14509","metadata_source":"pith","pith_arxiv_id":"2309.14509","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepSpeed Ulysses: System Optimizations for Enabling Training of Extreme Long Sequence Transformer Models","venue":"cs.LG","work_id":"bb119d0b-c7f0-412a-a426-f74bd6949a51","year":2023},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2309.14509","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:8cbdd7e751edb21207b2dbf835bad31f790b5906981ffb7f6f95aedb0c2340d1","observation_id":"11b0d50c-dfca-4844-9b5b-a3276475e058","resolution":{"observed_at":"2026-05-17T02:52:20.722917Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1705.06950","last_updated":"2017-05-19T12:07:01Z","snapshot_observed_at":"2026-08-08T17:46:50.107463Z","submitted_at":"2017-05-19T12:07:01Z","title":"The Kinetics Human Action Video Dataset","version":1},"cited_work":{"arxiv_id":"1705.06950","doi":null,"metadata_source":"pith","pith_arxiv_id":"1705.06950","snapshot_observed_at":"2026-07-09T11:16:11.423695Z","title":"The Kinetics Human Action Video Dataset","venue":"cs.CV","work_id":"c8a3de61-cfd3-4aeb-bcf7-a0372c015748","year":2017},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/1705.06950","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:1467f972311e6a9799521ffc3f74777c99f94e1cb36ad07abb37859cdbcc4c35","observation_id":"18c528c3-c6ae-451a-aef5-04f7daf09b2f","resolution":{"observed_at":"2026-05-17T02:52:20.727050Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03326","last_updated":"2024-10-26T16:35:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:59:44Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","version":3},"cited_work":{"arxiv_id":"2408.03326","doi":"10.48550/arxiv.2408.03326","metadata_source":"pith","pith_arxiv_id":"2408.03326","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","venue":"cs.CV","work_id":"f5f2452b-f2a9-49ac-b38d-c76e18cdfe49","year":2024},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2408.03326","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:eb4e5437855130871932cc02073c5e453eb9405051c7aca427b19b60dd8e4ba4","observation_id":"d67c6540-880e-4403-8880-9d79099455cb","resolution":{"observed_at":"2026-05-17T02:52:20.731060Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08418","last_updated":"2024-07-12T08:54:51Z","snapshot_observed_at":"2026-08-04T09:18:59.444343Z","submitted_at":"2024-06-12T17:01:04Z","title":"OmniCorpus: A Unified Multimodal Corpus of 10 Billion-Level Images Interleaved with Text","version":3},"cited_work":{"arxiv_id":"2406.08418","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08418","snapshot_observed_at":"2026-07-03T17:18:43.867707Z","title":"Omnicorpus: A unified multimodal corpus of 10 billion-level images interleaved with text","venue":null,"work_id":"7d9f48da-319d-4a13-8604-4f435eb9c224","year":2024},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2406.08418","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:68402f47c040ca386e57541f4cd1a45504483b5900ed4f2a5947e133bb65839a","observation_id":"f1b88b23-ddb9-408e-bb7e-bfcb08664a1d","resolution":{"observed_at":"2026-05-17T02:52:20.735829Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01889","last_updated":"2023-11-27T06:38:47Z","snapshot_observed_at":"2026-08-07T09:22:20.831075Z","submitted_at":"2023-10-03T08:44:50Z","title":"Ring Attention with Blockwise Transformers for Near-Infinite Context","version":4},"cited_work":{"arxiv_id":"2310.01889","doi":"10.48550/arxiv.2310.01889","metadata_source":"pith","pith_arxiv_id":"2310.01889","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Ring Attention with Blockwise Transformers for Near-Infinite Context","venue":"cs.CL","work_id":"982498c4-e80e-4e66-8c44-c60813be298d","year":2023},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2310.01889","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:74614e757e5b25a16ab7e02c06f7d7bbc4e99d333547bf6a186b742740a0e2de","observation_id":"c17e7587-fead-4bb2-a55f-bc474b366077","resolution":{"observed_at":"2026-05-17T02:52:20.740404Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-07-13T15:50:24.101224+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-13T15:50:24.101224+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.09418","last_updated":"2024-06-13T17:59:59Z","snapshot_observed_at":"2026-08-09T01:53:18.568676Z","submitted_at":"2024-06-13T17:59:59Z","title":"VideoGPT+: Integrating Image and Video Encoders for Enhanced Video Understanding","version":1},"cited_work":{"arxiv_id":"2406.09418","doi":"10.48550/arxiv.2406.09418","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.09418","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Videogpt+: Integrating image and video en- coders for enhanced video understanding","venue":"arXiv (Cornell University)","work_id":"b73f5b08-33ad-4759-b16b-0663cf9728b1","year":2024},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2406.09418","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:6f2af2c059bc6d5b371dd3fea02adb1faacf877ec88bdaf75f5bfea3b50d59ea","observation_id":"e17bb074-8708-42ef-85d6-ad5c89fc49cc","resolution":{"observed_at":"2026-05-17T02:52:20.745049Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":"2303.08774","doi":"10.1002/tea.20265","metadata_source":"pith","pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GPT-4 Technical Report","venue":"cs.CL","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","year":2023},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:e7b5bd0607352c458a587fefc781c703922d8f12346025d6f17c2c69abd0226a","observation_id":"3ed63893-240d-40cb-8740-2cf5cc93a0af","resolution":{"observed_at":"2026-05-17T02:52:20.749277Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00714","last_updated":"2024-10-28T16:37:57Z","snapshot_observed_at":"2026-07-06T18:55:41.459417Z","submitted_at":"2024-08-01T17:00:08Z","title":"SAM 2: Segment Anything in Images and Videos","version":2},"cited_work":{"arxiv_id":"2408.00714","doi":"10.1038/s41598-025-97590-3","metadata_source":"pith","pith_arxiv_id":"2408.00714","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SAM 2: Segment Anything in Images and Videos","venue":"cs.CV","work_id":"acc13f66-d814-44f9-9688-375688bf2d4a","year":2024},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2408.00714","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:1e2f9b74d30b126426661992727d71b3766de65d9ff1a30e526c9d2fb776e727","observation_id":"7757c5cc-869c-40b8-871e-e694fe99aeb6","resolution":{"observed_at":"2026-05-17T02:52:20.752983Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-24T04:24:23.885301+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-24T04:24:23.885301+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05530","last_updated":"2024-12-16T17:39:39Z","snapshot_observed_at":"2026-07-06T17:41:42.995949Z","submitted_at":"2024-03-08T18:54:20Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","version":5},"cited_work":{"arxiv_id":"2403.05530","doi":"10.48550/arxiv.2403.05530","metadata_source":"pith","pith_arxiv_id":"2403.05530","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","venue":"cs.CL","work_id":"80e3e977-f1bb-4c83-8d0c-1ab0a0c5c3f1","year":2024},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2403.05530","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:37a2c40b452616dcedb0625aff10a111d8808618f959000be1ebd31eca9fb3c0","observation_id":"e8d91143-00b3-4879-899c-184ca49812c0","resolution":{"observed_at":"2026-05-17T02:52:20.757255Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-02T03:08:14.426583+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T03:08:14.426583+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.02051","last_updated":"2024-03-28T12:41:14Z","snapshot_observed_at":"2026-08-10T13:36:40.776714Z","submitted_at":"2023-12-04T17:09:52Z","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","version":2},"cited_work":{"arxiv_id":"2312.02051","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.02051","snapshot_observed_at":"2026-07-04T06:39:37.643725Z","title":"Timechat: A time-sensitive multimodal large lan- guage model for long video understanding","venue":null,"work_id":"85fc1396-2fab-4238-9120-5878dc1705b6","year":2023},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2312.02051","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:53b284ff8c29195882e0c98c18d6d702c7f2fe1a43589f4e531765a621a70c2d","observation_id":"74733113-350f-4079-852c-d6a93c738232","resolution":{"observed_at":"2026-05-17T02:52:20.762202Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.17434","last_updated":"2024-10-22T21:21:37Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-22T21:21:37Z","title":"LongVU: Spatiotemporal Adaptive Compression for Long Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2410.17434","doi":"10.48550/arxiv.2410.17434","metadata_source":"pith","pith_arxiv_id":"2410.17434","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVU: Spatiotemporal Adaptive Compression for Long Video-Language Understanding","venue":"cs.CV","work_id":"82f3cbed-22bb-41b5-b9c9-67304154cd52","year":2024},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2410.17434","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:9bb7d123bc08c21366cd29a48a80894f18b72d6e918769a042c1eb72a6057a1a","observation_id":"790a7fd1-3f3b-4abd-829e-7bf975876128","resolution":{"observed_at":"2026-05-17T02:52:20.765897Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.14485","last_updated":"2024-12-10T12:45:31Z","snapshot_observed_at":"2026-08-10T13:02:21.561135Z","submitted_at":"2024-09-22T15:13:31Z","title":"Video-XL: Extra-Long Vision Language Model for Hour-Scale Video Understanding","version":4},"cited_work":{"arxiv_id":"2409.14485","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.14485","snapshot_observed_at":"2026-07-04T06:39:37.682244Z","title":"Video-XL: Extra-Long Vision Language Model for Hour-Scale Video Understanding","venue":null,"work_id":"1974d26a-10e2-4170-b1ee-626217ce7f74","year":2024},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2409.14485","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:5c403722f1063bbac1d4a068b880ba82b37bf39afee9575c99887a68566e3f2c","observation_id":"8467a30b-0658-47f3-8e17-081184e9bdde","resolution":{"observed_at":"2026-05-17T02:52:20.770114Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03290","last_updated":"2025-08-21T05:15:19Z","snapshot_observed_at":"2026-08-02T03:09:09.488305Z","submitted_at":"2024-10-04T10:04:37Z","title":"Grounded-VideoLLM: Sharpening Fine-grained Temporal Grounding in Video Large Language Models","version":2},"cited_work":{"arxiv_id":"2410.03290","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03290","snapshot_observed_at":"2026-07-04T15:09:55.249321Z","title":"Grounded-videollm: Sharpening fine-grained temporal grounding in video large language models","venue":null,"work_id":"6cda7c97-9a94-4872-b214-cddce67fea61","year":2024},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2410.03290","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:bfb06c5bb3e3eb6e766fed7d7b148465ae93aa0bd3b2b69fdbe9dd9468cb1200","observation_id":"dcaf2cd8-1f50-401c-83f5-de3f9366e583","resolution":{"observed_at":"2026-05-17T02:52:20.775068Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2409.02889","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T09:39:46.752891Z","title":"Longllava: Scaling multi-modal llms to 1000 images efficiently via hybrid architecture","venue":null,"work_id":"c93e268d-1703-46de-8123-15c73f06bc0b","year":2024},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:8aef549eff13b7660348790a76d330ad532375578e3446a322cdfd5d9068cf55","observation_id":"2cd6e254-0fd9-4e7c-bce5-86945bd97a24","resolution":{"observed_at":"2026-05-17T02:52:20.779477Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.06942","last_updated":"2024-01-04T05:00:34Z","snapshot_observed_at":"2026-07-06T15:53:46.393481Z","submitted_at":"2023-07-13T17:58:32Z","title":"InternVid: A Large-scale Video-Text Dataset for Multimodal Understanding and Generation","version":2},"cited_work":{"arxiv_id":"2307.06942","doi":null,"metadata_source":"pith","pith_arxiv_id":"2307.06942","snapshot_observed_at":"2026-07-04T06:39:37.513098Z","title":"InternVid: A Large-scale Video-Text Dataset for Multimodal Understanding and Generation","venue":"cs.CV","work_id":"e0787102-2f18-46ad-9660-6bfe466e3bbf","year":2023},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2307.06942","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:fe1dda7bc6dfe3e7dd4633099b927d4b14955b3468747ed43771ae557d525fd5","observation_id":"e8d8b3c2-c456-4eff-af15-513248c8a146","resolution":{"observed_at":"2026-05-17T02:52:20.783565Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.10228","last_updated":"2024-03-15T11:58:18Z","snapshot_observed_at":"2026-07-06T17:45:11.343569Z","submitted_at":"2024-03-15T11:58:18Z","title":"HawkEye: Training Video-Text LLMs for Grounding Text in Videos","version":1},"cited_work":{"arxiv_id":"2403.10228","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.10228","snapshot_observed_at":"2026-07-03T20:38:56.195732Z","title":"arXiv preprint arXiv:2403.10228 , year=","venue":null,"work_id":"fb2ec1aa-a3b0-43e2-9e93-2b73d3097074","year":2024},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2403.10228","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:230f7fd3005aaaed72ba32c3d72d7a5d6f56a026cb0355762e775fb49c5a6cd2","observation_id":"716f8125-2ca2-4693-9d43-eb94caa2de63","resolution":{"observed_at":"2026-05-17T02:52:20.788080Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:a24d7fa807b262646c10badef5cb4686c62f38fa8777c15222b0090f63ebf24a","observation_id":"56b398d2-a2a2-4efb-9779-f18fdb255c2f","resolution":{"observed_at":"2026-05-18T00:30:12.519784Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08394","last_updated":"2024-12-31T05:35:05Z","snapshot_observed_at":"2026-08-06T17:31:18.801416Z","submitted_at":"2024-06-12T16:44:50Z","title":"VisionLLM v2: An End-to-End Generalist Multimodal Large Language Model for Hundreds of Vision-Language Tasks","version":3},"cited_work":{"arxiv_id":"2406.08394","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08394","snapshot_observed_at":"2026-07-03T10:48:03.080928Z","title":"Visionllm v2: An end-to-end generalist multimodal large language model","venue":null,"work_id":"9f5a686c-3950-4ee0-b961-85f882649e32","year":2024},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2406.08394","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:6fc73722e25e84bf767b5178f2c49dd250b3ed0e1460670156865639339263ea","observation_id":"d5166116-14a6-4c38-b05d-f5764bbb5c74","resolution":{"observed_at":"2026-05-17T02:52:20.796481Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.10188","last_updated":"2024-12-13T02:32:06Z","snapshot_observed_at":"2026-08-05T14:57:53.592979Z","submitted_at":"2024-08-19T17:48:08Z","title":"LongVILA: Scaling Long-Context Visual Language Models for Long Videos","version":6},"cited_work":{"arxiv_id":"2408.10188","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.10188","snapshot_observed_at":"2026-07-03T10:48:02.935874Z","title":"LongVILA: Scaling Long-Context Visual Language Models for Long Videos","venue":"cs.CV","work_id":"3b739755-b666-4189-88cc-5d999bb41a9c","year":2024},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2408.10188","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:19871d900dfa787c3024fea9f3b49063fc39a1b9ce2582921935966dfa86423e","observation_id":"e2170d37-63ab-44c0-b67a-6e242c29aca4","resolution":{"observed_at":"2026-05-17T03:51:25.572615Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19326","last_updated":"2025-06-30T13:15:13Z","snapshot_observed_at":"2026-08-11T00:39:56.306346Z","submitted_at":"2024-12-26T18:56:05Z","title":"Task Preference Optimization: Improving Multimodal Large Language Models with Vision Task Alignment","version":2},"cited_work":{"arxiv_id":"2412.19326","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.19326","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Task preference optimization: Improving multimodal large language models with vision task alignment","venue":null,"work_id":"90adeb70-7a41-48c6-8e9e-e8346533501e","year":2024},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2412.19326","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:96c605e08670aa6aea4686f994212bcfe97e9d83a1effac2335e5e496ec4d71f","observation_id":"d9d4332a-5f1e-483f-a493-4b5634d2952e","resolution":{"observed_at":"2026-05-17T02:52:20.806469Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06040","last_updated":"2024-10-25T06:32:09Z","snapshot_observed_at":"2026-07-06T18:27:58.433967Z","submitted_at":"2024-06-10T06:17:55Z","title":"Vript: A Video Is Worth Thousands of Words","version":2},"cited_work":{"arxiv_id":"2406.06040","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.06040","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vript: A video is worth thousands of words","venue":null,"work_id":"c12e48fc-ded3-4345-957a-b3b9e47671f4","year":2024},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2406.06040","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:01ca2eb3e171ce3b072e31d058cac52c79604002e626f827da7270f1323f6d71","observation_id":"50d8304b-f818-4452-a6c0-04aa645e7471","resolution":{"observed_at":"2026-05-17T02:52:20.810354Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:16145e5c98eef6ad5f18477fae786d1df1c2e759cbad55614f6977534bb94bee","observation_id":"9710c9e3-ac62-487f-8521-303ae1e425ac","resolution":{"observed_at":"2026-05-20T06:20:36.894768Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04498","last_updated":"2023-12-18T12:15:26Z","snapshot_observed_at":"2026-08-06T14:33:43.511633Z","submitted_at":"2023-11-08T07:15:05Z","title":"NExT-Chat: An LMM for Chat, Detection and Segmentation","version":4},"cited_work":{"arxiv_id":"2311.04498","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04498","snapshot_observed_at":"2026-07-04T15:09:55.195600Z","title":"Next-chat: An lmm for chat, detection and segmentation","venue":null,"work_id":"67bdb7d8-4b20-47ce-91e4-3be89edf9028","year":2023},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2311.04498","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:c5d052c7c7e762257e891a73cff9b3c52dc0d3b0d076d806f3e0893ad99aea29","observation_id":"64f6bba2-b9e5-4be4-a078-cbdb17883279","resolution":{"observed_at":"2026-05-17T02:52:20.818808Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04817","last_updated":"2025-09-01T05:09:27Z","snapshot_observed_at":"2026-08-10T09:09:43.552999Z","submitted_at":"2023-12-08T03:33:38Z","title":"LvBench: A Benchmark for Long-form Video Understanding with Versatile Multi-modal Question Answering","version":2},"cited_work":{"arxiv_id":"2312.04817","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04817","snapshot_observed_at":"2026-07-03T10:48:03.110683Z","title":"Movqa: A benchmark of versatile question-answering for long-form movie understanding","venue":null,"work_id":"ecace4d0-749e-45f2-b598-0cfe1ccc1cc5","year":2023},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2312.04817","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:e3055d3c5dc60db05c81d4bc99387aef57b27ed748dc6cb350dd804dcd715657","observation_id":"957433df-c785-46cb-a36e-7d1aaf9c4f52","resolution":{"observed_at":"2026-05-17T02:52:20.823715Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04264","last_updated":"2025-01-01T15:53:58Z","snapshot_observed_at":"2026-08-03T20:38:36.602554Z","submitted_at":"2024-06-06T17:09:32Z","title":"MLVU: Benchmarking Multi-task Long Video Understanding","version":3},"cited_work":{"arxiv_id":"2406.04264","doi":"10.48550/arxiv.2406.04264","metadata_source":"pith","pith_arxiv_id":"2406.04264","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLVU: Benchmarking Multi-task Long Video Understanding","venue":"cs.CV","work_id":"346256da-dd21-4cc3-9a98-519467614854","year":2024},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2406.04264","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:4f4b4dadbf41d10310d3d023f2692a018e76b4ce3db7439d4298d5a81e95aa6f","observation_id":"c4101e25-5707-4f19-9c40-cc2457a97e13","resolution":{"observed_at":"2026-05-17T02:52:20.827990Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.10360","last_updated":"2024-12-13T18:53:24Z","snapshot_observed_at":"2026-08-06T08:59:28.938249Z","submitted_at":"2024-12-13T18:53:24Z","title":"Apollo: An Exploration of Video Understanding in Large Multimodal Models","version":1},"cited_work":{"arxiv_id":"2412.10360","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.10360","snapshot_observed_at":"2026-07-03T10:48:02.918791Z","title":"Apollo: An Exploration of Video Understanding in Large Multimodal Models","venue":null,"work_id":"a2514001-657a-44cc-8c33-0562df7ac008","year":2024},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2412.10360","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:a5779ac6a39a8b843fbbff6c9ebbbf78ad632ec3bed189aa663579dfef91d613","observation_id":"51649a7d-f4f2-4ec0-8e87-21456cbedbd9","resolution":{"observed_at":"2026-05-17T02:52:20.833571Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling"},"reference_resolution":{"displayed":37,"state_counts":{"malformed_identifier":0,"metadata_mismatch":5,"parse_uncertain":0,"unresolved":0,"verified_exact":32,"verified_fuzzy":0},"total_outbound_references":37},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 37 of 37 outbound references and 84 inbound Pith citation observations for arXiv:2501.12386."}