{"as_of":"2026-08-08T01:01:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:9205674f1d7162b325e79892be2a80b4c2ad09487a51326444eaca11d7dcc3f4","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":18,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":18,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":18,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":18,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:34:43.788047Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-02T12:26:57.184700Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":"2406.11303","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-07-02T12:26:57.184700Z","title":"Videovista: A versatile bench- mark for video understanding and reasoning","venue":null,"work_id":"b5b7e620-228d-4c25-9017-df5c7b62a760","year":2024},"citing_paper":{"arxiv_id":"2501.02955","last_updated":"2026-05-12T15:02:48Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-06T11:57:38Z","title":"MotionBench: Benchmarking and Improving Fine-grained Video Motion Understanding for Vision Language Models","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-23T05:44:31.546843Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2501.02955"},"observation_digest":"sha256:4ae0596d146f21d132f2f9f02e9259e8d0adde55ce242767601e9f24e63e79bc","observation_id":"df47b175-ed30-402b-9ad1-4104542abff3","resolution":{"observed_at":"2026-05-23T05:45:28.342097Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-08-07T15:34:43.788047Z","title":"Videovista: A versatile benchmark for video understanding and reasoning.arXiv preprint arXiv:2406.11303, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:43.788047Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:3cee0a0d7ab97f77aa63d6eb98ae8e03e159fe491573636e28df3d7bd3d0a5ae","observation_id":"29aaa453-09ca-4625-84b3-d9457014c874","resolution":{"observed_at":"2026-08-07T15:34:43.788047Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-08-07T14:03:01.131871Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20124","last_updated":"2025-05-27T12:10:27Z","snapshot_observed_at":"2026-08-07T13:56:23.519692Z","submitted_at":"2025-05-26T15:24:06Z","title":"TUNA: Comprehensive Fine-grained Temporal Understanding Evaluation on Dense Dynamic Videos","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-07T14:03:01.131871Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2505.20124"},"observation_digest":"sha256:b4ead16cadb939c26767bda71e8940621ea9f54bdb7919d531eb4ebe3bf2f977","observation_id":"fc66876d-c16b-4d4b-977f-7b397612597f","resolution":{"observed_at":"2026-08-07T14:03:01.131871Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-08-07T12:48:03.738031Z","title":"Videovista: A versatile benchmark for video understanding and reasoning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23484","last_updated":"2025-05-29T14:34:25Z","snapshot_observed_at":"2026-08-07T12:43:01.932268Z","submitted_at":"2025-05-29T14:34:25Z","title":"VCapsBench: A Large-scale Fine-grained Benchmark for Video Caption Quality Evaluation","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:03.738031Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2505.23484"},"observation_digest":"sha256:0be820351cad6635fe902b81a487673f926864352546d3b78954fde3a2a9cf94","observation_id":"7514bf63-6fd6-4d1f-91bb-5fac0532564c","resolution":{"observed_at":"2026-08-07T12:48:03.738031Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-08-07T12:45:46.866771Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23693","last_updated":"2025-05-29T17:31:13Z","snapshot_observed_at":"2026-08-07T18:41:28.769570Z","submitted_at":"2025-05-29T17:31:13Z","title":"VF-Eval: Evaluating Multimodal LLMs for Generating Feedback on AIGC Videos","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-07T12:45:46.866771Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2505.23693"},"observation_digest":"sha256:467cbe87d02b6cc6b989dbeed12a1553183c94b7e4e5d4ca02bed7ea9f8a58c9","observation_id":"f4e62474-1e58-4261-b56c-11dbe3e6481f","resolution":{"observed_at":"2026-08-07T12:45:46.866771Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-08-07T12:40:44.943565Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23922","last_updated":"2025-05-29T18:15:07Z","snapshot_observed_at":"2026-08-07T12:35:51.712975Z","submitted_at":"2025-05-29T18:15:07Z","title":"ScaleLong: A Multi-Timescale Benchmark for Long Video Understanding","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-07T12:40:44.943565Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2505.23922"},"observation_digest":"sha256:2f3dbb6b2764546f0dc34558c0daa6a77847b0593a0b7e8080ac3578982ca56b","observation_id":"6f5a593c-f7dc-4764-85f5-da4346e18514","resolution":{"observed_at":"2026-08-07T12:40:44.943565Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-08-07T12:32:10.535033Z","title":"Videovista: A versatile benchmark for video understanding and reasoning.arXiv preprint arXiv:2406.11303, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24346","last_updated":"2025-05-30T08:39:36Z","snapshot_observed_at":"2026-08-07T13:53:39.823176Z","submitted_at":"2025-05-30T08:39:36Z","title":"VUDG: A Dataset for Video Understanding Domain Generalization","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T12:32:10.535033Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2505.24346"},"observation_digest":"sha256:269d73b5cb79f5aff30b714114ba68f6493797f67ce5c89baf05f9d5614b2d07","observation_id":"408e7f61-f69b-46cd-97ec-da0311486d57","resolution":{"observed_at":"2026-08-07T12:32:10.535033Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":"2406.11303","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-07-02T12:26:57.184700Z","title":"Videovista: A versatile bench- mark for video understanding and reasoning","venue":null,"work_id":"b5b7e620-228d-4c25-9017-df5c7b62a760","year":2024},"citing_paper":{"arxiv_id":"2506.05425","last_updated":"2026-04-28T02:01:09Z","snapshot_observed_at":"2026-07-31T07:37:26.215945Z","submitted_at":"2025-06-05T05:51:35Z","title":"SIV-Bench: A Video Benchmark for Social Interaction Understanding and Reasoning","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-19T11:36:36.687324Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2506.05425"},"observation_digest":"sha256:4c0146633e12027ccbdbf36dccbeb2230d8cb7362b5b27ce35ddffe07b074a6a","observation_id":"f72b07e1-89ec-4e63-a23d-54c6a082e19c","resolution":{"observed_at":"2026-05-19T11:37:15.739001Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-08-07T04:22:55.901902Z","title":"Videovista: A versatile bench- mark for video understanding and reasoning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10857","last_updated":"2025-08-04T09:11:48Z","snapshot_observed_at":"2026-08-08T00:05:09.574699Z","submitted_at":"2025-06-12T16:17:17Z","title":"VRBench: A Benchmark for Multi-Step Reasoning in Long Narrative Videos","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T04:22:55.901902Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2506.10857"},"observation_digest":"sha256:a386aa000266a23e25102098fbd6563ddfaddb0680bf60f5a7dc15194ce350a3","observation_id":"72944cf2-74c2-4dd2-91fa-073fcf1672fa","resolution":{"observed_at":"2026-08-07T04:22:55.901902Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-08-07T00:41:51.290582Z","title":"Videovista: A versatile bench- mark for video understanding and reasoning.arXiv preprint arXiv:2406.11303, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12992","last_updated":"2025-06-15T23:20:08Z","snapshot_observed_at":"2026-08-07T00:35:03.776740Z","submitted_at":"2025-06-15T23:20:08Z","title":"SmartHome-Bench: A Comprehensive Benchmark for Video Anomaly Detection in Smart Homes Using Multi-Modal Large Language Models","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T00:41:51.290582Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2506.12992"},"observation_digest":"sha256:fd1bce31165cffae6bfe634ff7547a3efa91a9c10045c173120eb78ca4ddf664","observation_id":"fd3247c8-5278-4a34-86de-a9e8723a234c","resolution":{"observed_at":"2026-08-07T00:41:51.290582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-08-06T04:49:39.159644Z","title":"https://api.semanticscholar.org/CorpusID: 270559556","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.03039","last_updated":"2025-08-05T03:33:24Z","snapshot_observed_at":"2026-08-07T14:34:47.687957Z","submitted_at":"2025-08-05T03:33:24Z","title":"VideoForest: Person-Anchored Hierarchical Reasoning for Cross-Video Question Answering","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-06T04:49:39.159644Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2508.03039"},"observation_digest":"sha256:3a5dcff4c680904b003b261d4b8a4455848cda3aecdf2b2ae5874a74e76a9538","observation_id":"7d166b95-e89a-4822-9a73-8309863d4e2a","resolution":{"observed_at":"2026-08-06T04:49:39.159644Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-08-04T13:54:15.574763Z","title":", Chen , X","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.24563","last_updated":"2026-07-15T12:54:15Z","snapshot_observed_at":"2026-08-05T23:40:26.873053Z","submitted_at":"2025-09-29T10:16:05Z","title":"NeMo: Needle in a Montage for Video-Language Understanding","version":3},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-04T13:54:15.574763Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2509.24563"},"observation_digest":"sha256:43cadf40bfa2e74a246134040886d4cbe8571d159fc85afac96ec64fe437997b","observation_id":"4880c9fe-dc2e-4103-adad-2d91bd29d674","resolution":{"observed_at":"2026-08-04T13:54:15.574763Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":"2406.11303","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-07-02T12:26:57.184700Z","title":"Videovista: A versatile bench- mark for video understanding and reasoning","venue":null,"work_id":"b5b7e620-228d-4c25-9017-df5c7b62a760","year":2024},"citing_paper":{"arxiv_id":"2603.27259","last_updated":"2026-06-18T21:01:40Z","snapshot_observed_at":"2026-08-02T11:52:58.572026Z","submitted_at":"2026-03-28T12:44:19Z","title":"Seeing the Scene Matters: Revealing Forgetting in Video Understanding Models with a Scene-Aware Long-Video Benchmark","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-14T22:05:07.326202Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2603.27259"},"observation_digest":"sha256:ae7d1f64835b15005d627e895f1c35d75f84b610812992d81e9d01ee95102195","observation_id":"4f18c017-fa11-49b1-9ff5-5a40e6d0cc64","resolution":{"observed_at":"2026-05-14T22:08:04.406514Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":"2406.11303","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-07-02T12:26:57.184700Z","title":"Videovista: A versatile bench- mark for video understanding and reasoning","venue":null,"work_id":"b5b7e620-228d-4c25-9017-df5c7b62a760","year":2024},"citing_paper":{"arxiv_id":"2605.15342","last_updated":"2026-05-14T19:12:20Z","snapshot_observed_at":"2026-07-06T23:26:37.362117Z","submitted_at":"2026-05-14T19:12:20Z","title":"Minerva-Ego: Spatiotemporal Hints for Egocentric Video Understanding","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-19T16:02:53.887605Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2605.15342"},"observation_digest":"sha256:9f8616d833c74181743a7949b2620327baf51edb03baa2ddbab969ddc575e563","observation_id":"c02b375f-471e-495f-ba2d-dd8d1b12bdb0","resolution":{"observed_at":"2026-05-19T16:03:08.051487Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":"2406.11303","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-07-02T12:26:57.184700Z","title":"Videovista: A versatile bench- mark for video understanding and reasoning","venue":null,"work_id":"b5b7e620-228d-4c25-9017-df5c7b62a760","year":2024},"citing_paper":{"arxiv_id":"2606.03635","last_updated":"2026-06-02T13:31:57Z","snapshot_observed_at":"2026-08-05T07:22:25.874494Z","submitted_at":"2026-06-02T13:31:57Z","title":"VidMsg: A Benchmark for Implicit Message Inference in Short Videos","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-28T10:25:06.594946Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2606.03635"},"observation_digest":"sha256:52c0f28b4de4f2bb2b91ac0828c4f44f81d72f205d6562363da1f9ef7ff5a85c","observation_id":"b7b3ffe2-8116-4a87-b31b-d0d2b8287e22","resolution":{"observed_at":"2026-07-02T02:56:30.085934Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":"2406.11303","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-07-02T12:26:57.184700Z","title":"Videovista: A versatile bench- mark for video understanding and reasoning","venue":null,"work_id":"b5b7e620-228d-4c25-9017-df5c7b62a760","year":2024},"citing_paper":{"arxiv_id":"2606.06338","last_updated":"2026-06-04T16:12:43Z","snapshot_observed_at":"2026-07-06T23:46:10.612537Z","submitted_at":"2026-06-04T16:12:43Z","title":"StoryVideoQA: Scaling Deep Video Understanding with a Large-Scale, Multi-Genre and Auto-Generated Dataset","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-06-28T02:05:47.810096Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2606.06338"},"observation_digest":"sha256:b7022629017c07649d456e02256087c220d94830745b69a43aabbe6613f8b07f","observation_id":"fa3a6b1c-d6de-40ef-bfcd-2f77df218b05","resolution":{"observed_at":"2026-07-02T12:26:57.186187Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":"2406.11303","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-07-02T12:26:57.184700Z","title":"Videovista: A versatile bench- mark for video understanding and reasoning","venue":null,"work_id":"b5b7e620-228d-4c25-9017-df5c7b62a760","year":2024},"citing_paper":{"arxiv_id":"2606.28593","last_updated":"2026-06-26T20:38:04Z","snapshot_observed_at":"2026-07-07T00:02:38.226622Z","submitted_at":"2026-06-26T20:38:04Z","title":"Animation2Code: Evaluating Temporal Visual Reasoning in Video-to-Code Generation","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-30T00:53:52.724902Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2606.28593"},"observation_digest":"sha256:43f05678583070c4062f5ff4461933597d17fbf7b8bb67587f62a18c6d9f401f","observation_id":"7ad99eda-7280-48fb-a4c6-8582ee1d7c21","resolution":{"observed_at":"2026-07-01T16:05:49.750946Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-08-07T04:24:54.981358Z","title":"arXiv preprint arXiv:2406.11303 , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.06361","last_updated":"2026-08-06T17:57:06Z","snapshot_observed_at":"2026-08-08T00:16:29.921057Z","submitted_at":"2026-08-06T17:57:06Z","title":"The Low Frequency Trap: Video Language Models Fail at Simple Event Bookkeeping","version":1},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-08-07T04:24:54.981358Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2608.06361"},"observation_digest":"sha256:37719d870568713d95967a977c5da252783eb060ffefff4a6b1f65edc335b4eb","observation_id":"e059e240-3680-4a48-8baa-dbd830e98b72","resolution":{"observed_at":"2026-08-07T04:24:54.981358Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2406.11303/citation-record","integrity":"/paper/2406.11303/integrity","json":"/paper/2406.11303/citation-record.json","paper":"/paper/2406.11303"},"outbound":[],"paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 18 inbound Pith citation observations for arXiv:2406.11303."}