{"as_of":"2026-08-06T22:44:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:26674a378bf71ed8a89fcf7e2b45cfbf3ff1f23ee40c345a0c85a79d579f03a6","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":29,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":29,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":29,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":29,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T22:15:03.351183Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T00:29:15.346488Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2406.08035","last_updated":"2025-08-09T10:54:59Z","snapshot_observed_at":"2026-08-05T10:34:24.268925Z","submitted_at":"2024-06-12T09:36:52Z","title":"LVBench: An Extreme Long Video Understanding Benchmark","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-19T11:55:30.048525Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2406.08035"},"observation_digest":"sha256:606771f803975554d50b6faa4e6ebe75523ec6aced0c77b90bf76fdd7ee1d4c1","observation_id":"6afa325e-c4aa-4373-887a-6e3feca6923d","resolution":{"observed_at":"2026-05-19T11:55:30.200813Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2501.00574","last_updated":"2025-07-13T16:21:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-31T18:01:23Z","title":"VideoChat-Flash: Hierarchical Compression for Long-Context Video Modeling","version":4},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-18T04:02:43.261543Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2501.00574"},"observation_digest":"sha256:ada2d323dae673708c0a30652130b4c012b4a456e8dc4e01e09e7b489a8171be","observation_id":"fc2efdfa-89c7-4199-81f7-5ccd0e43f5b6","resolution":{"observed_at":"2026-05-18T04:02:43.448546Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2501.02955","last_updated":"2026-05-12T15:02:48Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-06T11:57:38Z","title":"MotionBench: Benchmarking and Improving Fine-grained Video Motion Understanding for Vision Language Models","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-23T05:44:31.546843Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2501.02955"},"observation_digest":"sha256:4153a856cf18cebf14442c04228bac1eb9d94701adfbefa106a319dc7b35cfd6","observation_id":"4c3f7916-40be-4917-86d3-699740523c97","resolution":{"observed_at":"2026-05-23T05:45:28.355335Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2501.13106","last_updated":"2025-06-03T03:33:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T18:59:46Z","title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","version":4},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-11T01:19:59.603343Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2501.13106"},"observation_digest":"sha256:861cdbd02b0ba3bcc1155b4af743b750bbef1310a3a0647cdcd2b7f41cdc2604","observation_id":"a75c095b-8f2e-4bbb-815c-15164036c109","resolution":{"observed_at":"2026-05-11T01:20:00.287485Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2503.21776","last_updated":"2025-10-22T16:42:24Z","snapshot_observed_at":"2026-08-05T07:15:29.998948Z","submitted_at":"2025-03-27T17:59:51Z","title":"Video-R1: Reinforcing Video Reasoning in MLLMs","version":4},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-12T09:43:00.208065Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2503.21776"},"observation_digest":"sha256:fc05d608bf15e5ec80d505449488090b56b80a98a62e9062c6c665fdc8732e0a","observation_id":"7b31e901-5b45-4c98-bc84-9819d5b33790","resolution":{"observed_at":"2026-05-12T09:43:00.408504Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2504.05299","last_updated":"2025-04-07T17:58:57Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-07T17:58:57Z","title":"SmolVLM: Redefining small and efficient multimodal models","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-13T20:23:50.552549Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2504.05299"},"observation_digest":"sha256:a3883e1f8b7e90e9b99c05dbd6b4e381f8684074905ff102aa8e41a13be43e04","observation_id":"3cd6b174-5f79-4bb2-b775-89284f17b9df","resolution":{"observed_at":"2026-05-13T20:23:51.734057Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2506.01844","last_updated":"2025-06-02T16:30:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-02T16:30:19Z","title":"SmolVLA: A Vision-Language-Action Model for Affordable and Efficient Robotics","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-11T21:22:36.902119Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2506.01844"},"observation_digest":"sha256:b5ea4445be3729349102a93d5829e45f2c511f8a792cdb3816bb3dfc4d549b70","observation_id":"c38d776f-7f12-467b-b91e-4bc3b14c4a9a","resolution":{"observed_at":"2026-05-11T21:22:37.480720Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-08-06T22:15:03.351183Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.22139","last_updated":"2025-07-22T07:42:31Z","snapshot_observed_at":"2026-08-06T22:07:43.493361Z","submitted_at":"2025-06-27T11:30:51Z","title":"Q-Frame: Query-aware Frame Selection and Multi-Resolution Adaptation for Video-LLMs","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T22:15:03.351183Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2506.22139"},"observation_digest":"sha256:4a437cf645a20083cb68b3c4243890da0f4409b546a1972aad01d0b23f9a5b2a","observation_id":"0f33d806-78f9-4ccf-9bd3-34caa259491c","resolution":{"observed_at":"2026-08-06T22:15:03.351183Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-08-06T21:37:05.338028Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-06T21:28:22.619337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:05.338028Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:4a173455a5e016e520853f94516aac398a06e007a8761ac5e6e3ecde5520d37a","observation_id":"0f4f6773-ee34-41bd-bd60-e24713ee2a90","resolution":{"observed_at":"2026-08-06T21:37:05.338028Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-08-06T14:05:32.407540Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19821","last_updated":"2025-08-02T13:22:34Z","snapshot_observed_at":"2026-08-06T14:05:31.386891Z","submitted_at":"2025-07-26T06:38:07Z","title":"LAVA: Language Driven Scalable and Versatile Traffic Video Analytics","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T14:05:32.407540Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2507.19821"},"observation_digest":"sha256:35f73b605607b3ae92bda82e48c59bc744d8bc3c4739d11ab5da8e90a1df337d","observation_id":"79cdbcc8-6c07-48b9-9e72-e300b01626e0","resolution":{"observed_at":"2026-08-06T14:05:32.407540Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-08-05T15:40:40.438516Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.19650","last_updated":"2025-08-29T02:25:23Z","snapshot_observed_at":"2026-08-05T15:40:38.625416Z","submitted_at":"2025-08-27T07:58:16Z","title":"Video-LevelGauge: Investigating Contextual Positional Bias in Large Video Language Models","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-05T15:40:40.438516Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2508.19650"},"observation_digest":"sha256:ccb7fc696d2beeb39502232b8ba0d977093a95616193e801de72a890df2ddc42","observation_id":"adce1126-a8a0-4236-b4ca-617e4f826596","resolution":{"observed_at":"2026-08-05T15:40:40.438516Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-08-05T15:10:16.769169Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.20478","last_updated":"2026-05-29T09:48:44Z","snapshot_observed_at":"2026-08-05T15:10:01.360889Z","submitted_at":"2025-08-28T06:55:08Z","title":"Video-MTR: Reinforced Multi-Turn Reasoning for Long Video Understanding","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-05T15:10:16.769169Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2508.20478"},"observation_digest":"sha256:23504f24180115a5f164684d26b6bdd2c651cc950b78b8f0a4e5da413ca55b01","observation_id":"83510f07-e3d2-4591-9f11-3089933d292e","resolution":{"observed_at":"2026-08-05T15:10:16.769169Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-08-04T21:27:34.988625Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.07680","last_updated":"2025-09-09T17:59:39Z","snapshot_observed_at":"2026-08-04T21:27:30.632810Z","submitted_at":"2025-09-09T17:59:39Z","title":"CAViAR: Critic-Augmented Video Agentic Reasoning","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-04T21:27:34.988625Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2509.07680"},"observation_digest":"sha256:527b069523c021bba0d42b7ca00d8b6ad3e81e444104ea4ff8e73ffc30cd284c","observation_id":"9faf2417-4896-46ee-9883-69e0bdcc3e56","resolution":{"observed_at":"2026-08-04T21:27:34.988625Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:7fe2d707b94367b5eed3a2e596ed84d935d979932b5e6d673311cd9fb3a1c640","observation_id":"b5c68ae3-7955-4fa6-b102-f464b8aa660a","resolution":{"observed_at":"2026-05-17T02:51:28.132505Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2601.14724","last_updated":"2026-05-07T12:10:26Z","snapshot_observed_at":"2026-08-05T03:02:47.238051Z","submitted_at":"2026-01-21T07:26:15Z","title":"HERMES: KV Cache as Hierarchical Memory for Efficient Streaming Video Understanding","version":4},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-16T12:55:04.564442Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2601.14724"},"observation_digest":"sha256:0a7070a00bbd4aea72d6bee43c1a28322af0227147aac261f86e244040ea1aa6","observation_id":"8313bc12-2f58-443d-87b3-565b183979ff","resolution":{"observed_at":"2026-05-16T12:57:53.873910Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2604.08120","last_updated":"2026-04-09T11:40:25Z","snapshot_observed_at":"2026-07-06T22:57:18.202215Z","submitted_at":"2026-04-09T11:40:25Z","title":"Small Vision-Language Models are Smart Compressors for Long Video Understanding","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:49.654073Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2604.08120"},"observation_digest":"sha256:cad53682520112eb1587e63d5c12f3ddb499da0ff7f9f5cc4187dfe86c39de0f","observation_id":"d1c7aabf-cc5c-4e9e-adc7-4a15863bba87","resolution":{"observed_at":"2026-05-11T00:15:51.479344Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2604.14149","last_updated":"2026-04-16T15:48:38Z","snapshot_observed_at":"2026-07-06T23:02:00.082783Z","submitted_at":"2026-04-15T17:59:52Z","title":"One Token per Highly Selective Frame: Towards Extreme Compression for Long Video Understanding","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-10T13:28:58.920442Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2604.14149"},"observation_digest":"sha256:90d6bbadc63042fc97c2052dcdcad988476561a787e5df44bb75f26efc5bd168","observation_id":"82579acb-6fab-499a-978b-dcefacfa5f2e","resolution":{"observed_at":"2026-05-10T13:30:26.569125Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2604.17052","last_updated":"2026-04-18T16:22:05Z","snapshot_observed_at":"2026-08-02T06:02:58.067734Z","submitted_at":"2026-04-18T16:22:05Z","title":"OASIS: On-Demand Hierarchical Event Memory for Streaming Video Reasoning","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-10T06:51:52.861981Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2604.17052"},"observation_digest":"sha256:b7e1221d9197e1e2908627b8ea4f9f0d02f73b8f1a10c877903023c9364b6916","observation_id":"3e04ced6-a4e9-4ed8-b2f9-66302f455250","resolution":{"observed_at":"2026-05-10T06:56:47.759603Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2605.00444","last_updated":"2026-05-01T06:24:40Z","snapshot_observed_at":"2026-07-06T23:13:52.304925Z","submitted_at":"2026-05-01T06:24:40Z","title":"Scaling Video Understanding via Compact Latent Multi-Agent Collaboration","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-09T20:11:11.410051Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2605.00444"},"observation_digest":"sha256:569b37c54ec08f044f8d7fcdea113114d71f5314950ea41b75592de019d12c6f","observation_id":"92803c71-da0a-4e87-a538-1536a6e706fd","resolution":{"observed_at":"2026-05-11T15:21:09.522561Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2605.07575","last_updated":"2026-05-11T11:58:53Z","snapshot_observed_at":"2026-07-06T23:19:56.415523Z","submitted_at":"2026-05-08T10:46:10Z","title":"Response-G1: Explicit Scene Graph Modeling for Proactive Streaming Video Understanding","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-05-11T02:30:55.939351Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2605.07575"},"observation_digest":"sha256:ab2d71f88dee6937b20e1a013fcf2d5f60ddd71c379160e1a05e4b2ed75302cc","observation_id":"c8418107-8250-4579-88e0-92a470464e44","resolution":{"observed_at":"2026-05-11T03:20:56.479634Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2605.07575","last_updated":"2026-05-11T11:58:53Z","snapshot_observed_at":"2026-07-06T23:19:56.415523Z","submitted_at":"2026-05-08T10:46:10Z","title":"Response-G1: Explicit Scene Graph Modeling for Proactive Streaming Video Understanding","version":2},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-05-12T03:00:34.728880Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2605.07575"},"observation_digest":"sha256:7a43561e957cdbbedce95b4d356e0d2982d8c2e5ef5aa4b9c5221f40af5144ea","observation_id":"c1005e51-57e9-4c04-9b1d-862c7d6bfbde","resolution":{"observed_at":"2026-05-12T03:01:17.700487Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2605.17921","last_updated":"2026-06-01T06:29:58Z","snapshot_observed_at":"2026-07-06T23:28:50.117638Z","submitted_at":"2026-05-18T06:29:44Z","title":"An Efficient Streaming Video Understanding Framework with Agentic Control","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-05-20T11:30:22.151045Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2605.17921"},"observation_digest":"sha256:de78220cee7a50b4414eed35f5ebbd8c381db0799eb1c7c081fd239f9548eebb","observation_id":"88c56c0f-3dab-484a-878e-e22f4a3402a5","resolution":{"observed_at":"2026-05-20T11:33:14.452976Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2605.25621","last_updated":"2026-05-25T09:23:19Z","snapshot_observed_at":"2026-07-06T23:35:36.158984Z","submitted_at":"2026-05-25T09:23:19Z","title":"StreamOV: Streaming Omni-Video Understanding via Evidence-Guided Memory and Response Triggering","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-06-29T22:27:17.092553Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2605.25621"},"observation_digest":"sha256:560defd9225ee5f40afe003ea0e7b05b515a586557666d04ca27a6a9b22a34d5","observation_id":"06199d9c-9c90-4ba0-bca6-6dfab8cdebb0","resolution":{"observed_at":"2026-06-29T22:34:02.071338Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2606.12195","last_updated":"2026-06-10T15:17:08Z","snapshot_observed_at":"2026-08-01T02:09:41.655807Z","submitted_at":"2026-06-10T15:17:08Z","title":"InternVideo3: Agentify Foundation Models with Multimodal Contextual Reasoning","version":1},"reference_index":296,"source":"arxiv_source","source_observed_at":"2026-06-27T09:48:27.652901Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2606.12195"},"observation_digest":"sha256:15ad2b4fc0fac80d053ea74cacc9ae55bc61b3dd6b150fe74cd21bfc1d700f6d","observation_id":"4f5b8170-f9d8-43a5-82cc-fa3598a11b04","resolution":{"observed_at":"2026-07-03T10:48:03.192666Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2606.18441","last_updated":"2026-06-16T19:42:54Z","snapshot_observed_at":"2026-08-03T17:38:18.858871Z","submitted_at":"2026-06-16T19:42:54Z","title":"Reasoning as Intersection: Consensus-Frame Alignment for Visual Focus in Video-MLLMs","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-06-27T01:02:37.632461Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2606.18441"},"observation_digest":"sha256:f8f9a7e6b1436d6fe7fa784b0155d363a30b54f917a8e442276661f85e4e3961","observation_id":"ed14d6ca-5e62-4cb0-9eed-d7f32f883594","resolution":{"observed_at":"2026-07-03T20:58:57.726244Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2606.19341","last_updated":"2026-07-17T18:52:18Z","snapshot_observed_at":"2026-08-06T12:20:24.096310Z","submitted_at":"2026-06-17T17:59:56Z","title":"Native Active Perception as Reasoning for Omni-Modal Understanding","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-26T21:14:11.297384Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2606.19341"},"observation_digest":"sha256:031b1b856ffab52a890f62bb7cc1bbf350bab88ec7499a3c57ce82a50998d643","observation_id":"a605ff3f-0bea-437f-a1a5-671a2adbaa57","resolution":{"observed_at":"2026-07-04T00:29:15.349201Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-08-02T10:57:19.810078Z","title":"Video-LLaV A: Learning united visual representation by alignment before projection","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.19341","last_updated":"2026-07-17T18:52:18Z","snapshot_observed_at":"2026-08-06T12:20:24.096310Z","submitted_at":"2026-06-17T17:59:56Z","title":"Native Active Perception as Reasoning for Omni-Modal Understanding","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-02T10:57:19.810078Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2606.19341"},"observation_digest":"sha256:caf2fe7b29bfd2fecc6b6efb628bb3b33ae2fee173e988d5f6b8023cbe9594d0","observation_id":"71114075-e508-49b7-8ba5-8ebae0751e93","resolution":{"observed_at":"2026-08-02T10:57:19.810078Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-12T09:09:15.248815Z","title":"Kangaroo: A powerful video-language model supporting long-context video input.arXiv preprint arXiv:2408.15542, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.02607","last_updated":"2026-07-01T13:51:02Z","snapshot_observed_at":"2026-07-12T09:09:14.587474Z","submitted_at":"2026-07-01T13:51:02Z","title":"Latent Visual Cache for Video Reasoning","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-07-12T09:09:15.248815Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2607.02607"},"observation_digest":"sha256:9a43bc18773e4a0e4f6e49728c90f0da15e5ba318a096860b4ae99e1fba51d59","observation_id":"01d453a6-8134-402f-ba94-496cc1c508df","resolution":{"observed_at":"2026-07-12T09:09:15.248815Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-08-01T08:49:13.271024Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.21013","last_updated":"2026-07-25T07:42:56Z","snapshot_observed_at":"2026-08-06T15:47:58.632965Z","submitted_at":"2026-07-23T07:52:40Z","title":"EmoAgent-R1: Towards Multimodal Emotion Understanding with Reinforcement Learning-based Dynamic Agent Specialization","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-01T08:49:13.271024Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2607.21013"},"observation_digest":"sha256:fc97a03bcc5ddd0d76b26beabfaccb04e638b8b210c3378b4a1c2ecc46e29724","observation_id":"30cccac3-b3ac-40f9-adef-3c640cb6616b","resolution":{"observed_at":"2026-08-01T08:49:13.271024Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2408.15542/citation-record","integrity":"/paper/2408.15542/integrity","json":"/paper/2408.15542/citation-record.json","paper":"/paper/2408.15542"},"outbound":[],"paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 29 inbound Pith citation observations for arXiv:2408.15542."}