{"as_of":"2026-08-07T22:06:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:092a4efd22a6c0075baccba4c34cc7f3dcaf91bdb6cbcaf6eca2ea2e76efa257","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":35,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":35,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":35,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":35,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:02:29.351108Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T06:39:37.518508Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2404.16821","last_updated":"2024-04-29T20:24:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-25T17:59:19Z","title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","version":2},"reference_index":120,"source":"pdf_text","source_observed_at":"2026-05-12T20:58:58.849040Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2404.16821"},"observation_digest":"sha256:8724b32b2b9a6dad0e77d34d6d5368a2cf72ed83a2630cec81f9f0bb89990eb9","observation_id":"0ea28090-2a9b-430f-a743-afc02aa2e0ea","resolution":{"observed_at":"2026-05-12T20:58:59.244814Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2410.05363","last_updated":"2024-10-07T17:56:04Z","snapshot_observed_at":"2026-07-06T19:29:14.335016Z","submitted_at":"2024-10-07T17:56:04Z","title":"Towards World Simulator: Crafting Physical Commonsense-Based Benchmark for Video Generation","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-05-18T14:39:59.870039Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2410.05363"},"observation_digest":"sha256:d55fcdc904cc943a553ef946f12bb50e083f6e983e5c3e8ca06c17c53400840d","observation_id":"6482212f-2fea-4ee8-af46-15f76d04a960","resolution":{"observed_at":"2026-05-18T14:40:00.050951Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2410.17434","last_updated":"2024-10-22T21:21:37Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-22T21:21:37Z","title":"LongVU: Spatiotemporal Adaptive Compression for Long Video-Language Understanding","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-16T13:53:33.585035Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2410.17434"},"observation_digest":"sha256:54040070ebec1d84caba7330878a7b1362d21b9b39db4277de70317e3553c800","observation_id":"67648cd9-4a98-42e9-9df2-3427496945b5","resolution":{"observed_at":"2026-05-16T13:53:33.707536Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2412.05271","last_updated":"2025-09-26T12:52:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-06T18:57:08Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","version":5},"reference_index":257,"source":"pdf_text","source_observed_at":"2026-05-10T13:23:57.588851Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2412.05271"},"observation_digest":"sha256:08724d7992b8d71cc747c25603c6b4e70127d718670495d4e10172a4551bbf78","observation_id":"be600ca6-544c-4333-806e-450eb4add321","resolution":{"observed_at":"2026-05-10T13:23:58.143903Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2412.15689","last_updated":"2026-05-06T21:36:42Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-20T09:07:36Z","title":"DOLLAR: Few-Step Video Generation via Distillation and Latent Reward Optimization","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-23T06:57:50.897865Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2412.15689"},"observation_digest":"sha256:e48b8f2df1f12b97231025ef65fabb78f8941d579a78d1c13f89db0fe2cc1cea","observation_id":"9b3a2c62-41ad-4638-849f-bff192cadec6","resolution":{"observed_at":"2026-05-23T07:02:41.738013Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2501.02955","last_updated":"2026-05-12T15:02:48Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-06T11:57:38Z","title":"MotionBench: Benchmarking and Improving Fine-grained Video Motion Understanding for Vision Language Models","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-23T05:44:31.546843Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2501.02955"},"observation_digest":"sha256:e47631e3916b9ef13262522b8c8bb26835128a2b50a7a00514653f2fb7bf9a68","observation_id":"d4ddbc8d-88bb-4485-b83d-33783b4f2547","resolution":{"observed_at":"2026-05-23T05:45:28.358964Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2501.05067","last_updated":"2026-04-20T07:42:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-09T08:43:57Z","title":"LLaVA-Octopus: Unlocking Instruction-Driven Adaptive Projector Fusion for Video Understanding","version":3},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-05-23T06:01:00.775721Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2501.05067"},"observation_digest":"sha256:b5eded5531c7cc550ef792fc31a64a931aac0466eafbd394002f77591a23c38e","observation_id":"e76c4e79-5a2b-4e07-83b2-a35350833ea4","resolution":{"observed_at":"2026-05-23T06:02:37.554003Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2501.13106","last_updated":"2025-06-03T03:33:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T18:59:46Z","title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","version":4},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-11T01:19:59.603343Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2501.13106"},"observation_digest":"sha256:6eb6d26770baf145ecc8a06bca900ef4aff6f9eb3e5312361d7331cfccdd47d2","observation_id":"fe182bdf-a1eb-452c-ab36-07d2a8315a47","resolution":{"observed_at":"2026-05-11T01:19:59.742495Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-08-07T15:02:29.351108Z","title":"Internvideo2: Scaling video foundation models for multimodal video understanding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16594","last_updated":"2025-05-22T12:28:50Z","snapshot_observed_at":"2026-08-07T14:55:54.546955Z","submitted_at":"2025-05-22T12:28:50Z","title":"Temporal Object Captioning for Street Scene Videos from LiDAR Tracks","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T15:02:29.351108Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2505.16594"},"observation_digest":"sha256:872c9052b31d36d03802e8428566be57e2f8e74dc9a54f3027de042ff5f60e5c","observation_id":"b85f535a-1e35-4405-ac04-20df1c5043b4","resolution":{"observed_at":"2026-08-07T15:02:29.351108Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-08-07T14:24:10.374002Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19125","last_updated":"2025-05-25T12:44:12Z","snapshot_observed_at":"2026-08-07T14:17:46.430576Z","submitted_at":"2025-05-25T12:44:12Z","title":"RTime-QA: A Benchmark for Atomic Temporal Event Understanding in Large Multi-modal Models","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:10.374002Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2505.19125"},"observation_digest":"sha256:4d002236dffff103671c41638b403d6b4f334c51884eaf5e7188d8d2a5608aae","observation_id":"fbc88a1a-335d-41cc-9183-f0aab1822dcd","resolution":{"observed_at":"2026-08-07T14:24:10.374002Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-08-07T13:52:47.260706Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20644","last_updated":"2025-05-27T02:45:14Z","snapshot_observed_at":"2026-08-07T13:47:38.739847Z","submitted_at":"2025-05-27T02:45:14Z","title":"HCQA-1.5 @ Ego4D EgoSchema Challenge 2025","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T13:52:47.260706Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2505.20644"},"observation_digest":"sha256:4ab28944f055b866fef4de41cbb46b4be4e63f926a8ad13740d0392572c1b8c6","observation_id":"4f82f8a4-75f2-4b0b-bd9f-12cdf4e124f5","resolution":{"observed_at":"2026-08-07T13:52:47.260706Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-08-07T13:48:15.401308Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding.arXiv preprint arXiv:2403.15377, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20920","last_updated":"2025-05-27T09:10:59Z","snapshot_observed_at":"2026-08-07T13:41:10.441862Z","submitted_at":"2025-05-27T09:10:59Z","title":"HuMoCon: Concept Discovery for Human Motion Understanding","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-07T13:48:15.401308Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2505.20920"},"observation_digest":"sha256:4a287db5af7607e79a3e807e65c079bb85e6452a1975842582c74cf2f8ec71cb","observation_id":"c51851bd-3b04-4f8f-9f48-14366c20f7cd","resolution":{"observed_at":"2026-08-07T13:48:15.401308Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-08-07T10:27:05.499105Z","title":"Internvideo2: Scaling video foundation models for multimodal video understanding.arXiv preprint arXiv:2403.15377, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05328","last_updated":"2025-07-22T07:00:35Z","snapshot_observed_at":"2026-08-07T11:49:44.536475Z","submitted_at":"2025-06-05T17:58:33Z","title":"AV-Reasoner: Improving and Benchmarking Clue-Grounded Audio-Visual Counting for MLLMs","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T10:27:05.499105Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2506.05328"},"observation_digest":"sha256:858e9d3cc92b761111fbbf50ced8ff89c2404e91e073fe303a1e41486a7353b7","observation_id":"5c096c6d-ac30-465c-a225-b79619168202","resolution":{"observed_at":"2026-08-07T10:27:05.499105Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-08-07T10:27:30.926631Z","title":"Internvideo2: Scaling video foundation models for multimodal video understanding.arXiv preprint arXiv:2403.15377, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05336","last_updated":"2025-07-05T11:38:26Z","snapshot_observed_at":"2026-08-07T10:19:00.643126Z","submitted_at":"2025-06-05T17:59:29Z","title":"VideoMolmo: Spatio-Temporal Grounding Meets Pointing","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T10:27:30.926631Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2506.05336"},"observation_digest":"sha256:48ff2c361a779e261d9f1f4bb40a0f36f1a4fd67b73cbc4af22a662ce11bb3eb","observation_id":"34c1518b-6a96-4e58-ad3a-81998148f13c","resolution":{"observed_at":"2026-08-07T10:27:30.926631Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-08-07T04:07:59.061572Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.11659","last_updated":"2025-06-13T10:40:23Z","snapshot_observed_at":"2026-08-07T04:01:57.468738Z","submitted_at":"2025-06-13T10:40:23Z","title":"An Empirical study on LLM-based Log Retrieval for Software Engineering Metadata Management","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T04:07:59.061572Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2506.11659"},"observation_digest":"sha256:14aa15c556426f645a233a9cd60390fe0f36892902c10a0c85f72e514b4f9f74","observation_id":"a206936a-a89b-4446-b0ca-5da22eb7bbec","resolution":{"observed_at":"2026-08-07T04:07:59.061572Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-08-07T00:51:56.851307Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12585","last_updated":"2025-06-14T17:39:03Z","snapshot_observed_at":"2026-08-07T00:43:16.069936Z","submitted_at":"2025-06-14T17:39:03Z","title":"DejaVid: Encoder-Agnostic Learned Temporal Matching for Video Classification","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T00:51:56.851307Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2506.12585"},"observation_digest":"sha256:67e30f9c6fd56ebce5b2072fb08d1c4b1e8075db34e8b18b14e4f4e4c05c1d98","observation_id":"30471e61-9c6b-4718-a4bb-197fbf6e0fd0","resolution":{"observed_at":"2026-08-07T00:51:56.851307Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-08-07T00:26:00.161920Z","title":"InternVideo2: Scaling foundation models for multimodal video understanding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14356","last_updated":"2025-06-17T09:51:51Z","snapshot_observed_at":"2026-08-07T00:15:38.945528Z","submitted_at":"2025-06-17T09:51:51Z","title":"EVA02-AT: Egocentric Video-Language Understanding with Spatial-Temporal Rotary Positional Embeddings and Symmetric Optimization","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T00:26:00.161920Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2506.14356"},"observation_digest":"sha256:281c054908cf0595cf93ff32839100527bf12c5763f4f696ff8b53dcd1e02669","observation_id":"0c6c2b82-1304-4e98-8301-7975bacc3446","resolution":{"observed_at":"2026-08-07T00:26:00.161920Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-08-06T23:42:05.964749Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.16450","last_updated":"2025-06-19T16:35:49Z","snapshot_observed_at":"2026-08-06T23:36:03.391214Z","submitted_at":"2025-06-19T16:35:49Z","title":"How Far Can Off-the-Shelf Multimodal Large Language Models Go in Online Episodic Memory Question Answering?","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T23:42:05.964749Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2506.16450"},"observation_digest":"sha256:a65556fbbecb0ab62234072fbb9496e3041890b5566eeccae140d67067a639cb","observation_id":"26244bf5-0094-4a2f-a175-3efb65d83ac2","resolution":{"observed_at":"2026-08-06T23:42:05.964749Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-08-06T22:24:33.036457Z","title":"Internvideo2: Scaling video foundation models for multimodal video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.21862","last_updated":"2025-06-27T02:29:58Z","snapshot_observed_at":"2026-08-07T08:24:31.697101Z","submitted_at":"2025-06-27T02:29:58Z","title":"LLaVA-Scissor: Token Compression with Semantic Connected Components for Video LLMs","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-06T22:24:33.036457Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2506.21862"},"observation_digest":"sha256:8e7fda9c30c79331f34d19c480295da3a4689f3ddb9bda7efc809b50f831c576","observation_id":"f2293dc2-fba5-4555-8a7b-352c33d71e45","resolution":{"observed_at":"2026-08-06T22:24:33.036457Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2507.04590","last_updated":"2025-07-07T00:51:57Z","snapshot_observed_at":"2026-08-02T08:05:44.477432Z","submitted_at":"2025-07-07T00:51:57Z","title":"VLM2Vec-V2: Advancing Multimodal Embedding for Videos, Images, and Visual Documents","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-18T14:10:14.929207Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2507.04590"},"observation_digest":"sha256:eb9cea1162790ac514f26dd2fb0b6a735f996987679f6eab476f54ee5ba81aa3","observation_id":"e683b406-530d-4883-8e6f-29049fa30f24","resolution":{"observed_at":"2026-05-18T14:10:15.130757Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-08-06T17:24:58.979468Z","title":"Internvideo2: Scaling video foundation models for multimodal video understanding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10960","last_updated":"2025-07-15T03:42:14Z","snapshot_observed_at":"2026-08-06T17:17:47.352213Z","submitted_at":"2025-07-15T03:42:14Z","title":"Whom to Respond To? A Transformer-Based Model for Multi-Party Social Robot Interaction","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T17:24:58.979468Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2507.10960"},"observation_digest":"sha256:85964703e778ed59f70dd14c6a4817c9e8cb1c136e9ab8f1059ca83b830c76ae","observation_id":"3f95cab1-e444-4ba3-b563-c49f4618a8e4","resolution":{"observed_at":"2026-08-06T17:24:58.979468Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-08-06T13:57:04.588143Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding.arXiv preprint arXiv:2403.15377, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19924","last_updated":"2025-08-01T12:25:21Z","snapshot_observed_at":"2026-08-07T14:18:30.406157Z","submitted_at":"2025-07-26T12:03:47Z","title":"HumanSAM: Classifying Human-centric Forgery Videos in Human Spatial, Appearance, and Motion Anomaly","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:04.588143Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2507.19924"},"observation_digest":"sha256:b0a8800f3d8104afea45639044e77d2cae59588dcb4e7c0e59b5a57c57f2d824","observation_id":"49664a06-07a5-49c3-a551-aeaea6f87f3f","resolution":{"observed_at":"2026-08-06T13:57:04.588143Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-08-04T20:20:36.886005Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding.arXiv preprint arXiv:2403.15377, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.08621","last_updated":"2025-09-10T14:17:53Z","snapshot_observed_at":"2026-08-06T13:15:27.312512Z","submitted_at":"2025-09-10T14:17:53Z","title":"AdsQA: Towards Advertisement Video Understanding","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-04T20:20:36.886005Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2509.08621"},"observation_digest":"sha256:25cc51f59761afc3fe5b35c2fe25765e8d9001268ddc9d4d2e1c111b65d32aaa","observation_id":"63b3f5bf-aa89-4242-8139-ecca80d15c9e","resolution":{"observed_at":"2026-08-04T20:20:36.886005Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2510.09608","last_updated":"2025-10-10T17:59:58Z","snapshot_observed_at":"2026-07-06T22:32:22.953630Z","submitted_at":"2025-10-10T17:59:58Z","title":"StreamingVLM: Real-Time Understanding for Infinite Video Streams","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-17T11:51:33.345812Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2510.09608"},"observation_digest":"sha256:4d8bf8a69f998430af1c9005c8602664996bed26f803e9799ec5182baa222076","observation_id":"e4d6e6dc-1985-459c-a7f7-9e3438c90669","resolution":{"observed_at":"2026-05-17T11:51:33.436812Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2604.02891","last_updated":"2026-04-03T09:00:38Z","snapshot_observed_at":"2026-07-06T22:52:10.923214Z","submitted_at":"2026-04-03T09:00:38Z","title":"Progressive Video Condensation with MLLM Agent for Long-form Video Understanding","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-13T20:40:41.380829Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2604.02891"},"observation_digest":"sha256:f7e47f80a4dbb55e403698c1c8d2ffa719ca612492a53ea7e32243b54cf78569","observation_id":"b4991dee-1341-458f-860d-3b3bdb084ac4","resolution":{"observed_at":"2026-05-13T20:43:14.848489Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2605.02834","last_updated":"2026-05-05T09:59:53Z","snapshot_observed_at":"2026-07-06T23:15:51.008483Z","submitted_at":"2026-05-04T17:11:16Z","title":"VideoNet: A Large-Scale Dataset for Domain-Specific Action Recognition","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-08T18:35:48.379198Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2605.02834"},"observation_digest":"sha256:8a62b157265e1c55ff9a91a5f1e16ae55f733a4b3d57ee25332cc6ab5adc78be","observation_id":"9d8ecc9e-0f2c-4a83-b438-04eb4b7f16cc","resolution":{"observed_at":"2026-05-09T06:20:41.219911Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2605.12954","last_updated":"2026-05-13T03:40:21Z","snapshot_observed_at":"2026-07-06T23:24:32.412966Z","submitted_at":"2026-05-13T03:40:21Z","title":"AdaFocus: Adaptive Relevance-Diversity Sampling with Zero-Cache Look-back for Efficient Long Video Understanding","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-14T19:43:29.123615Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2605.12954"},"observation_digest":"sha256:255ac54d897bfb6a29f81780d29641a29af6ac783283f13ed887924ffffa3531","observation_id":"c27af26f-fadf-4167-a8b4-78c00b2f34a4","resolution":{"observed_at":"2026-05-14T19:47:53.818862Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2605.16403","last_updated":"2026-05-13T05:00:19Z","snapshot_observed_at":"2026-07-06T23:27:34.514541Z","submitted_at":"2026-05-13T05:00:19Z","title":"When Vision Speaks for Sound","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-20T22:12:52.160596Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2605.16403"},"observation_digest":"sha256:e922f4467c5908253ffba667462f04e445ff7ebf9b046a56c9ef4767698ba89e","observation_id":"3d550601-08e8-4fef-bd0a-d450e29c01e8","resolution":{"observed_at":"2026-05-20T22:13:46.852107Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2606.00775","last_updated":"2026-05-30T15:40:00Z","snapshot_observed_at":"2026-07-06T23:41:24.911421Z","submitted_at":"2026-05-30T15:40:00Z","title":"GIRL-DETR: Gradient-Isolated Reinforcement Learning for Video Moment Retrieval","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-06-28T19:12:19.056273Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2606.00775"},"observation_digest":"sha256:755ee13786bb2fa9f3dcf0d6ee713ec4ba63a5fe2c9ee1df066d221aeaffb0a0","observation_id":"825cb563-f5c1-4de2-8225-143668a3cd13","resolution":{"observed_at":"2026-06-28T19:12:34.375723Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2606.03635","last_updated":"2026-06-02T13:31:57Z","snapshot_observed_at":"2026-08-05T07:22:25.874494Z","submitted_at":"2026-06-02T13:31:57Z","title":"VidMsg: A Benchmark for Implicit Message Inference in Short Videos","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-28T10:25:06.594946Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2606.03635"},"observation_digest":"sha256:455282ffdeafe58474b79ef5cadf60405f7b773adc9972f375e6b0dc1bf26cbb","observation_id":"d949fff9-4f47-4bd9-9828-14cf0d8d1e52","resolution":{"observed_at":"2026-07-02T02:56:30.063976Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2606.06249","last_updated":"2026-06-04T14:52:12Z","snapshot_observed_at":"2026-07-06T23:46:05.829177Z","submitted_at":"2026-06-04T14:52:12Z","title":"GRAMformer: Any-Order Modality Interactions via Volumetric Multimodal Cross-Attention","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-28T02:22:03.908592Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2606.06249"},"observation_digest":"sha256:a75168ede8973b1acbb661a7fe9e414d0e6e6b95c89bbf1976e2b9c5d93258bd","observation_id":"d76c4a66-6383-49dc-b854-8df5bd1deb6e","resolution":{"observed_at":"2026-07-02T12:06:56.379175Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2606.21734","last_updated":"2026-06-19T20:43:49Z","snapshot_observed_at":"2026-08-05T18:05:51.515234Z","submitted_at":"2026-06-19T20:43:49Z","title":"HPP: Hierarchical Programmatic Probing for Long Video Understanding by Decoupling Perception and Reasoning","version":1},"reference_index":91,"source":"arxiv_source","source_observed_at":"2026-06-26T14:19:53.450263Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2606.21734"},"observation_digest":"sha256:28247c690aafa12462feb4e05dc6dbf457fed3bd4b551805c291b6a0886fe450","observation_id":"f705bd2b-15de-4e9e-a1c3-f31dd6f8147c","resolution":{"observed_at":"2026-07-04T06:39:37.519902Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":"2403.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-04T06:39:37.518508Z","title":"Internvideo2: Scaling video foundation mod- els for multimodal video understanding","venue":null,"work_id":"1f0f626c-f5dc-428e-9eb7-3e99516c1cc5","year":2024},"citing_paper":{"arxiv_id":"2606.28215","last_updated":"2026-06-26T16:05:58Z","snapshot_observed_at":"2026-08-07T08:50:21.361337Z","submitted_at":"2026-06-26T16:05:58Z","title":"HAT-4D: Lifting Monocular Video for 4D Multi-Object Interactions via Human-Agent Collaboration","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-06-29T04:18:02.341742Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2606.28215"},"observation_digest":"sha256:9f0a0202a0fd518e304fba9a30f68a3b24e4a494d9e48dfc9ef4826da38b4b2a","observation_id":"32b0cf74-2c11-4ecd-993b-6de5dc48818f","resolution":{"observed_at":"2026-07-01T17:05:50.568899Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-07-12T01:50:59.184754Z","title":"arXiv preprint arXiv:2403.15377 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.03530","last_updated":"2026-07-03T17:59:58Z","snapshot_observed_at":"2026-08-04T18:44:44.436706Z","submitted_at":"2026-07-03T17:59:58Z","title":"MentalThink: Shaping Thoughts in Mental SVG World","version":1},"reference_index":141,"source":"arxiv_source","source_observed_at":"2026-07-12T01:50:59.184754Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2607.03530"},"observation_digest":"sha256:c542c9e6e5c05c3cb35f59ba4e15b4dff443f4740f44726e92cec408d65dfd04","observation_id":"f73a5272-7603-4603-9b33-02aa4aa36cda","resolution":{"observed_at":"2026-07-12T01:50:59.184754Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.15377","snapshot_observed_at":"2026-08-01T17:45:04.287534Z","title":"arXiv preprint arXiv:2403.15377 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.17560","last_updated":"2026-07-20T05:09:36Z","snapshot_observed_at":"2026-08-05T13:38:45.921055Z","submitted_at":"2026-07-20T05:09:36Z","title":"Reinforcement Learning: From Algorithms To Foundation Models","version":1},"reference_index":84,"source":"arxiv_source","source_observed_at":"2026-08-01T17:45:04.287534Z"},"links":{"cited_paper":"/paper/2403.15377","citing_paper":"/paper/2607.17560"},"observation_digest":"sha256:5fd9d4334188b05fa5af66e34a6b0b59533c7fb4928d83c5e516ae63c888b043","observation_id":"a5be6167-3f15-4bf3-8fd8-5b7bbee536d4","resolution":{"observed_at":"2026-08-01T17:45:04.287534Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2403.15377/citation-record","integrity":"/paper/2403.15377/integrity","json":"/paper/2403.15377/citation-record.json","paper":"/paper/2403.15377"},"outbound":[],"paper":{"arxiv_id":"2403.15377","last_updated":"2024-08-14T14:31:50Z","latest_version":4,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-01T19:17:08.239976Z","submitted_at":"2024-03-22T17:57:42Z","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 35 inbound Pith citation observations for arXiv:2403.15377."}