{"as_of":"2026-08-09T13:45:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:94203ed49b856734e2e69cc1ab813b5e178c40c41d57f0cd5a16070fb1f48f85","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":27,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":27,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":27,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":27,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T17:37:38.120634Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T06:39:37.643725Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2312.02051","last_updated":"2024-03-28T12:41:14Z","snapshot_observed_at":"2026-07-06T16:56:42.232555Z","submitted_at":"2023-12-04T17:09:52Z","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","version":2},"cited_work":{"arxiv_id":"2312.02051","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.02051","snapshot_observed_at":"2026-07-04T06:39:37.643725Z","title":"Timechat: A time-sensitive multimodal large lan- guage model for long video understanding","venue":null,"work_id":"85fc1396-2fab-4238-9120-5878dc1705b6","year":2023},"citing_paper":{"arxiv_id":"2403.00476","last_updated":"2024-06-03T04:13:39Z","snapshot_observed_at":"2026-08-04T21:17:37.211833Z","submitted_at":"2024-03-01T12:02:19Z","title":"TempCompass: Do Video LLMs Really Understand Videos?","version":3},"reference_index":116,"source":"arxiv_source","source_observed_at":"2026-05-17T02:46:16.632743Z"},"links":{"cited_paper":"/paper/2312.02051","citing_paper":"/paper/2403.00476"},"observation_digest":"sha256:6de9bd923dca86da51d56c4e773a6f8c3d1cbb01b932f387142ac87b54ba184c","observation_id":"2894a04d-1d74-4417-a654-57b721d1551d","resolution":{"observed_at":"2026-05-17T02:46:16.807356Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.02051","last_updated":"2024-03-28T12:41:14Z","snapshot_observed_at":"2026-07-06T16:56:42.232555Z","submitted_at":"2023-12-04T17:09:52Z","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","version":2},"cited_work":{"arxiv_id":"2312.02051","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.02051","snapshot_observed_at":"2026-07-04T06:39:37.643725Z","title":"Timechat: A time-sensitive multimodal large lan- guage model for long video understanding","venue":null,"work_id":"85fc1396-2fab-4238-9120-5878dc1705b6","year":2023},"citing_paper":{"arxiv_id":"2406.04264","last_updated":"2025-01-01T15:53:58Z","snapshot_observed_at":"2026-08-03T20:38:36.602554Z","submitted_at":"2024-06-06T17:09:32Z","title":"MLVU: Benchmarking Multi-task Long Video Understanding","version":3},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-14T19:55:26.333923Z"},"links":{"cited_paper":"/paper/2312.02051","citing_paper":"/paper/2406.04264"},"observation_digest":"sha256:cb78d9a10ef531b651e9a7e334911f3fa3029927e6431e5a896e02ea7358f49d","observation_id":"cd0c9261-42f4-4beb-8ad3-e83f25ecae50","resolution":{"observed_at":"2026-05-14T19:55:26.427758Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.02051","last_updated":"2024-03-28T12:41:14Z","snapshot_observed_at":"2026-07-06T16:56:42.232555Z","submitted_at":"2023-12-04T17:09:52Z","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","version":2},"cited_work":{"arxiv_id":"2312.02051","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.02051","snapshot_observed_at":"2026-07-04T06:39:37.643725Z","title":"Timechat: A time-sensitive multimodal large lan- guage model for long video understanding","venue":null,"work_id":"85fc1396-2fab-4238-9120-5878dc1705b6","year":2023},"citing_paper":{"arxiv_id":"2406.07476","last_updated":"2024-10-30T06:49:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-11T17:22:23Z","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","version":3},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-11T02:44:53.284345Z"},"links":{"cited_paper":"/paper/2312.02051","citing_paper":"/paper/2406.07476"},"observation_digest":"sha256:c8cb36f74e8e0a4dde870a9acf08fb9c5f15f86aa56fee6a25d5520abe037852","observation_id":"e1fd15f6-ca78-464e-9243-f4693104d822","resolution":{"observed_at":"2026-05-11T02:44:53.620986Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.02051","last_updated":"2024-03-28T12:41:14Z","snapshot_observed_at":"2026-07-06T16:56:42.232555Z","submitted_at":"2023-12-04T17:09:52Z","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","version":2},"cited_work":{"arxiv_id":"2312.02051","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.02051","snapshot_observed_at":"2026-07-04T06:39:37.643725Z","title":"Timechat: A time-sensitive multimodal large lan- guage model for long video understanding","venue":null,"work_id":"85fc1396-2fab-4238-9120-5878dc1705b6","year":2023},"citing_paper":{"arxiv_id":"2406.08035","last_updated":"2025-08-09T10:54:59Z","snapshot_observed_at":"2026-08-08T19:38:26.415599Z","submitted_at":"2024-06-12T09:36:52Z","title":"LVBench: An Extreme Long Video Understanding Benchmark","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-19T11:55:30.048525Z"},"links":{"cited_paper":"/paper/2312.02051","citing_paper":"/paper/2406.08035"},"observation_digest":"sha256:5bfa4ef6812e66e8ee8a1f675b494b6399e54ef94d5336d1ed47528d77041ba9","observation_id":"ed7db905-7503-481c-b1d0-c2d6a37bc33b","resolution":{"observed_at":"2026-05-19T11:55:30.099872Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.02051","last_updated":"2024-03-28T12:41:14Z","snapshot_observed_at":"2026-07-06T16:56:42.232555Z","submitted_at":"2023-12-04T17:09:52Z","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","version":2},"cited_work":{"arxiv_id":"2312.02051","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.02051","snapshot_observed_at":"2026-07-04T06:39:37.643725Z","title":"Timechat: A time-sensitive multimodal large lan- guage model for long video understanding","venue":null,"work_id":"85fc1396-2fab-4238-9120-5878dc1705b6","year":2023},"citing_paper":{"arxiv_id":"2408.16500","last_updated":"2024-08-29T12:59:12Z","snapshot_observed_at":"2026-08-05T11:54:14.447608Z","submitted_at":"2024-08-29T12:59:12Z","title":"CogVLM2: Visual Language Models for Image and Video Understanding","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-05-16T20:10:27.633010Z"},"links":{"cited_paper":"/paper/2312.02051","citing_paper":"/paper/2408.16500"},"observation_digest":"sha256:bbcd27829d1b49e22290a2ff77db68a3332daa6db970feb434ab85de46b1e90b","observation_id":"ca324a57-d643-4590-a8bb-7f477c6345fe","resolution":{"observed_at":"2026-05-16T20:10:27.687282Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.02051","last_updated":"2024-03-28T12:41:14Z","snapshot_observed_at":"2026-07-06T16:56:42.232555Z","submitted_at":"2023-12-04T17:09:52Z","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","version":2},"cited_work":{"arxiv_id":"2312.02051","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.02051","snapshot_observed_at":"2026-07-04T06:39:37.643725Z","title":"Timechat: A time-sensitive multimodal large lan- guage model for long video understanding","venue":null,"work_id":"85fc1396-2fab-4238-9120-5878dc1705b6","year":2023},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2312.02051","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:c3c6f39a3add51d83b271721e5d360edbf8965e45a2c6f8d7d2167fcb926114b","observation_id":"74733113-350f-4079-852c-d6a93c738232","resolution":{"observed_at":"2026-05-17T02:52:20.762202Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.02051","last_updated":"2024-03-28T12:41:14Z","snapshot_observed_at":"2026-07-06T16:56:42.232555Z","submitted_at":"2023-12-04T17:09:52Z","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.02051","snapshot_observed_at":"2026-08-08T17:37:38.120634Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.05887","last_updated":"2025-02-09T13:00:53Z","snapshot_observed_at":"2026-08-09T04:00:39.010576Z","submitted_at":"2025-02-09T13:00:53Z","title":"MTPChat: A Multimodal Time-Aware Persona Dataset for Conversational Agents","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-08T17:37:38.120634Z"},"links":{"cited_paper":"/paper/2312.02051","citing_paper":"/paper/2502.05887"},"observation_digest":"sha256:13a7ced027a12afc2c8882d16d746552239c1750781db662c94d6e5caad26d85","observation_id":"b7bb7727-ba7d-4ecc-ba1e-6079cc4e9f21","resolution":{"observed_at":"2026-08-08T17:37:38.120634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.02051","last_updated":"2024-03-28T12:41:14Z","snapshot_observed_at":"2026-07-06T16:56:42.232555Z","submitted_at":"2023-12-04T17:09:52Z","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.02051","snapshot_observed_at":"2026-08-07T13:08:06.501876Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.22613","last_updated":"2025-05-28T17:29:34Z","snapshot_observed_at":"2026-08-07T13:00:45.454980Z","submitted_at":"2025-05-28T17:29:34Z","title":"RICO: Improving Accuracy and Completeness in Image Recaptioning via Visual Reconstruction","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-07T13:08:06.501876Z"},"links":{"cited_paper":"/paper/2312.02051","citing_paper":"/paper/2505.22613"},"observation_digest":"sha256:f53e437ac9912274c124e6689eff460a4d6db3fbaa42615f370633ed61067973","observation_id":"0bbf7527-2e0a-43dd-8862-846ec2812208","resolution":{"observed_at":"2026-08-07T13:08:06.501876Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.02051","last_updated":"2024-03-28T12:41:14Z","snapshot_observed_at":"2026-07-06T16:56:42.232555Z","submitted_at":"2023-12-04T17:09:52Z","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.02051","snapshot_observed_at":"2026-08-07T10:17:17.607966Z","title":"Timechat: A time-sensitive multimodal large language model for long video understanding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.05787","last_updated":"2025-08-04T22:48:35Z","snapshot_observed_at":"2026-08-07T10:11:02.494841Z","submitted_at":"2025-06-06T06:33:16Z","title":"EASG-Bench: Video Q&A Benchmark with Egocentric Action Scene Graphs","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T10:17:17.607966Z"},"links":{"cited_paper":"/paper/2312.02051","citing_paper":"/paper/2506.05787"},"observation_digest":"sha256:8215695c7ff573ca61fe33e4de98ef16edda74eb22407db1cdf73db788cc2481","observation_id":"db37397a-99a4-4b89-a1ef-8f05e1286840","resolution":{"observed_at":"2026-08-07T10:17:17.607966Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.02051","last_updated":"2024-03-28T12:41:14Z","snapshot_observed_at":"2026-07-06T16:56:42.232555Z","submitted_at":"2023-12-04T17:09:52Z","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.02051","snapshot_observed_at":"2026-08-06T22:24:31.342995Z","title":"Timechat: A time-sensitive multimodal large language model for long video understanding.ArXiv, abs/2312.02051, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.21862","last_updated":"2025-06-27T02:29:58Z","snapshot_observed_at":"2026-08-09T10:29:21.701927Z","submitted_at":"2025-06-27T02:29:58Z","title":"LLaVA-Scissor: Token Compression with Semantic Connected Components for Video LLMs","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-06T22:24:31.342995Z"},"links":{"cited_paper":"/paper/2312.02051","citing_paper":"/paper/2506.21862"},"observation_digest":"sha256:cb6be6b41377204228e7fefd79909e6e347cb4d3079c72c47605c166ebf6e1cc","observation_id":"039f19b2-59e5-41e2-b0c1-42a3e1a0c856","resolution":{"observed_at":"2026-08-06T22:24:31.342995Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.02051","last_updated":"2024-03-28T12:41:14Z","snapshot_observed_at":"2026-07-06T16:56:42.232555Z","submitted_at":"2023-12-04T17:09:52Z","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.02051","snapshot_observed_at":"2026-08-06T22:20:12.971888Z","title":"Timechat: A time-sensitive multimodal large language model for long video understanding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.21891","last_updated":"2025-06-27T04:05:12Z","snapshot_observed_at":"2026-08-06T22:14:20.472000Z","submitted_at":"2025-06-27T04:05:12Z","title":"DIVE: Deep-search Iterative Video Exploration A Technical Report for the CVRR Challenge at CVPR 2025","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T22:20:12.971888Z"},"links":{"cited_paper":"/paper/2312.02051","citing_paper":"/paper/2506.21891"},"observation_digest":"sha256:ce76b91b2ee86066cf08ac9b784a575d801a323040fdd07385708bd75e8cd166","observation_id":"c76d8af5-6ecf-405e-8499-77ad6f20ccc0","resolution":{"observed_at":"2026-08-06T22:20:12.971888Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.02051","last_updated":"2024-03-28T12:41:14Z","snapshot_observed_at":"2026-07-06T16:56:42.232555Z","submitted_at":"2023-12-04T17:09:52Z","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.02051","snapshot_observed_at":"2026-08-06T22:00:06.968000Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.00068","last_updated":"2025-06-28T12:12:06Z","snapshot_observed_at":"2026-08-09T10:41:05.778908Z","submitted_at":"2025-06-28T12:12:06Z","title":"MANTA: Cross-Modal Semantic Alignment and Information-Theoretic Optimization for Long-form Multimodal Understanding","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-06T22:00:06.968000Z"},"links":{"cited_paper":"/paper/2312.02051","citing_paper":"/paper/2507.00068"},"observation_digest":"sha256:1b9005572c5fb7a9464b1732d59c11f6d304fa8e91ad75e7cf873349467ed027","observation_id":"ce5d04b7-30d0-4adc-8fce-0d5913fe9b60","resolution":{"observed_at":"2026-08-06T22:00:06.968000Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.02051","last_updated":"2024-03-28T12:41:14Z","snapshot_observed_at":"2026-07-06T16:56:42.232555Z","submitted_at":"2023-12-04T17:09:52Z","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","version":2},"cited_work":{"arxiv_id":"2312.02051","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.02051","snapshot_observed_at":"2026-07-04T06:39:37.643725Z","title":"Timechat: A time-sensitive multimodal large lan- guage model for long video understanding","venue":null,"work_id":"85fc1396-2fab-4238-9120-5878dc1705b6","year":2023},"citing_paper":{"arxiv_id":"2604.02860","last_updated":"2026-04-03T08:26:12Z","snapshot_observed_at":"2026-07-06T22:52:10.923214Z","submitted_at":"2026-04-03T08:26:12Z","title":"A Paradigm Shift: Fully End-to-End Training for Temporal Sentence Grounding in Videos","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-13T19:43:57.298338Z"},"links":{"cited_paper":"/paper/2312.02051","citing_paper":"/paper/2604.02860"},"observation_digest":"sha256:d575c47f1e81861b324788ad5986722f08e319e08c078b567624f468e3617d6f","observation_id":"b85caa48-9723-4b62-a11a-6e2c148e0057","resolution":{"observed_at":"2026-05-13T19:48:11.525080Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.02051","last_updated":"2024-03-28T12:41:14Z","snapshot_observed_at":"2026-07-06T16:56:42.232555Z","submitted_at":"2023-12-04T17:09:52Z","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","version":2},"cited_work":{"arxiv_id":"2312.02051","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.02051","snapshot_observed_at":"2026-07-04T06:39:37.643725Z","title":"Timechat: A time-sensitive multimodal large lan- guage model for long video understanding","venue":null,"work_id":"85fc1396-2fab-4238-9120-5878dc1705b6","year":2023},"citing_paper":{"arxiv_id":"2604.25886","last_updated":"2026-06-16T05:05:15Z","snapshot_observed_at":"2026-08-08T01:22:27.799880Z","submitted_at":"2026-04-28T17:29:19Z","title":"MarkIt: Training-Free Visual Markers for Precise Video Temporal Grounding","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-07T14:10:27.416341Z"},"links":{"cited_paper":"/paper/2312.02051","citing_paper":"/paper/2604.25886"},"observation_digest":"sha256:3f450bbd2abed5dd31e21625cfc1351bd34f811be804fd312d0340941d01f2c0","observation_id":"bbc5f630-013a-40d7-a1a7-2ec06f18f519","resolution":{"observed_at":"2026-05-12T00:46:13.738281Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.02051","last_updated":"2024-03-28T12:41:14Z","snapshot_observed_at":"2026-07-06T16:56:42.232555Z","submitted_at":"2023-12-04T17:09:52Z","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","version":2},"cited_work":{"arxiv_id":"2312.02051","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.02051","snapshot_observed_at":"2026-07-04T06:39:37.643725Z","title":"Timechat: A time-sensitive multimodal large lan- guage model for long video understanding","venue":null,"work_id":"85fc1396-2fab-4238-9120-5878dc1705b6","year":2023},"citing_paper":{"arxiv_id":"2604.25886","last_updated":"2026-06-16T05:05:15Z","snapshot_observed_at":"2026-08-08T01:22:27.799880Z","submitted_at":"2026-04-28T17:29:19Z","title":"MarkIt: Training-Free Visual Markers for Precise Video Temporal Grounding","version":3},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-07-01T08:41:31.700367Z"},"links":{"cited_paper":"/paper/2312.02051","citing_paper":"/paper/2604.25886"},"observation_digest":"sha256:f4d594adfc01f66056d1cd10494da755092fa247268af58ad850563e158a339b","observation_id":"7ba36f90-8d5b-4224-860b-09a8bac66510","resolution":{"observed_at":"2026-07-01T08:45:34.755963Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.02051","last_updated":"2024-03-28T12:41:14Z","snapshot_observed_at":"2026-07-06T16:56:42.232555Z","submitted_at":"2023-12-04T17:09:52Z","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","version":2},"cited_work":{"arxiv_id":"2312.02051","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.02051","snapshot_observed_at":"2026-07-04T06:39:37.643725Z","title":"Timechat: A time-sensitive multimodal large lan- guage model for long video understanding","venue":null,"work_id":"85fc1396-2fab-4238-9120-5878dc1705b6","year":2023},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2312.02051","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:cb12bebb08784dd38d07412a90218db7ad9f84d8dcc905f2ee042dee114e558e","observation_id":"62853dd5-b767-49ba-a9a9-4a13f37bcfab","resolution":{"observed_at":"2026-06-29T22:13:59.539308Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.02051","last_updated":"2024-03-28T12:41:14Z","snapshot_observed_at":"2026-07-06T16:56:42.232555Z","submitted_at":"2023-12-04T17:09:52Z","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","version":2},"cited_work":{"arxiv_id":"2312.02051","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.02051","snapshot_observed_at":"2026-07-04T06:39:37.643725Z","title":"Timechat: A time-sensitive multimodal large lan- guage model for long video understanding","venue":null,"work_id":"85fc1396-2fab-4238-9120-5878dc1705b6","year":2023},"citing_paper":{"arxiv_id":"2606.05736","last_updated":"2026-06-04T05:55:15Z","snapshot_observed_at":"2026-07-06T23:45:42.379051Z","submitted_at":"2026-06-04T05:55:15Z","title":"VTI-CoT: Visual-Textual Interleaved Chain of Thought for Video Reasoning","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-28T01:52:44.785582Z"},"links":{"cited_paper":"/paper/2312.02051","citing_paper":"/paper/2606.05736"},"observation_digest":"sha256:fbdbe2be42e796f0e1b008e4c80bd4b6259c9df6d3203805625c5a742213d8ac","observation_id":"c5036575-fc9c-419e-84c4-37101201c4e5","resolution":{"observed_at":"2026-07-02T12:46:56.832350Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.02051","last_updated":"2024-03-28T12:41:14Z","snapshot_observed_at":"2026-07-06T16:56:42.232555Z","submitted_at":"2023-12-04T17:09:52Z","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","version":2},"cited_work":{"arxiv_id":"2312.02051","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.02051","snapshot_observed_at":"2026-07-04T06:39:37.643725Z","title":"Timechat: A time-sensitive multimodal large lan- guage model for long video understanding","venue":null,"work_id":"85fc1396-2fab-4238-9120-5878dc1705b6","year":2023},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2312.02051","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:23105a4967e3bd8163c8b67de8cc3e261a497156c25e4766a2790f67aaf72bfe","observation_id":"99f19cce-399a-45ea-af49-7a991abf30fc","resolution":{"observed_at":"2026-07-02T17:27:15.504136Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.02051","last_updated":"2024-03-28T12:41:14Z","snapshot_observed_at":"2026-07-06T16:56:42.232555Z","submitted_at":"2023-12-04T17:09:52Z","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","version":2},"cited_work":{"arxiv_id":"2312.02051","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.02051","snapshot_observed_at":"2026-07-04T06:39:37.643725Z","title":"Timechat: A time-sensitive multimodal large lan- guage model for long video understanding","venue":null,"work_id":"85fc1396-2fab-4238-9120-5878dc1705b6","year":2023},"citing_paper":{"arxiv_id":"2606.21734","last_updated":"2026-06-19T20:43:49Z","snapshot_observed_at":"2026-08-05T18:05:51.515234Z","submitted_at":"2026-06-19T20:43:49Z","title":"HPP: Hierarchical Programmatic Probing for Long Video Understanding by Decoupling Perception and Reasoning","version":1},"reference_index":193,"source":"arxiv_source","source_observed_at":"2026-06-26T14:19:53.450263Z"},"links":{"cited_paper":"/paper/2312.02051","citing_paper":"/paper/2606.21734"},"observation_digest":"sha256:254611c4d03af7f52a02e1248083a9b7c613befb16e3d6e6a91db9484facd2f8","observation_id":"720ab0c3-e66d-454c-a5d5-e85fc57d5822","resolution":{"observed_at":"2026-07-04T06:39:37.645127Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.02051","last_updated":"2024-03-28T12:41:14Z","snapshot_observed_at":"2026-07-06T16:56:42.232555Z","submitted_at":"2023-12-04T17:09:52Z","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","version":2},"cited_work":{"arxiv_id":"2312.02051","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.02051","snapshot_observed_at":"2026-07-04T06:39:37.643725Z","title":"Timechat: A time-sensitive multimodal large lan- guage model for long video understanding","venue":null,"work_id":"85fc1396-2fab-4238-9120-5878dc1705b6","year":2023},"citing_paper":{"arxiv_id":"2607.00867","last_updated":"2026-07-15T08:03:34Z","snapshot_observed_at":"2026-08-02T09:17:20.227144Z","submitted_at":"2026-07-01T12:32:29Z","title":"EFlow: Learning Evidence Flow for Long-Video Reasoning with Adaptive Reflection","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-02T14:19:04.890074Z"},"links":{"cited_paper":"/paper/2312.02051","citing_paper":"/paper/2607.00867"},"observation_digest":"sha256:b30666e7d38bec139c2be1dd389076dcccb7a42514c8eae1a4456054c8b937c7","observation_id":"3e1094fb-58b5-43cf-8056-6f2b0e2139b4","resolution":{"observed_at":"2026-07-02T14:27:03.324975Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.02051","last_updated":"2024-03-28T12:41:14Z","snapshot_observed_at":"2026-07-06T16:56:42.232555Z","submitted_at":"2023-12-04T17:09:52Z","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.02051","snapshot_observed_at":"2026-07-12T09:13:38.446621Z","title":"Timo Schick, Jane Dwivedi-Yu, Roberto Dessì, Roberta Raileanu, Maria Lomeli, Luke Zettlemoyer, Nicola Cancedda, and Thomas Scialom","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.00867","last_updated":"2026-07-15T08:03:34Z","snapshot_observed_at":"2026-08-02T09:17:20.227144Z","submitted_at":"2026-07-01T12:32:29Z","title":"EFlow: Learning Evidence Flow for Long-Video Reasoning with Adaptive Reflection","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-07-12T09:13:38.446621Z"},"links":{"cited_paper":"/paper/2312.02051","citing_paper":"/paper/2607.00867"},"observation_digest":"sha256:133f51a099f031ddfa20d79535a302d246f657adf56e6a8ac26452b507d23ac5","observation_id":"af988246-9c7f-4120-a332-cdde69102010","resolution":{"observed_at":"2026-07-12T09:13:38.446621Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.02051","last_updated":"2024-03-28T12:41:14Z","snapshot_observed_at":"2026-07-06T16:56:42.232555Z","submitted_at":"2023-12-04T17:09:52Z","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.02051","snapshot_observed_at":"2026-08-02T09:17:20.566842Z","title":"Timo Schick, Jane Dwivedi-Yu, Roberto Dessì, Roberta Raileanu, Maria Lomeli, Luke Zettlemoyer, Nicola Cancedda, and Thomas Scialom","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.00867","last_updated":"2026-07-15T08:03:34Z","snapshot_observed_at":"2026-08-02T09:17:20.227144Z","submitted_at":"2026-07-01T12:32:29Z","title":"EFlow: Learning Evidence Flow for Long-Video Reasoning with Adaptive Reflection","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-02T09:17:20.566842Z"},"links":{"cited_paper":"/paper/2312.02051","citing_paper":"/paper/2607.00867"},"observation_digest":"sha256:7e22d19e099cf14b7ee0cb5cd5283047674b41c318cf5097b95c58cbdb9d4236","observation_id":"66d9599d-a935-4146-8499-6ec150eea4e5","resolution":{"observed_at":"2026-08-02T09:17:20.566842Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.02051","last_updated":"2024-03-28T12:41:14Z","snapshot_observed_at":"2026-07-06T16:56:42.232555Z","submitted_at":"2023-12-04T17:09:52Z","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.02051","snapshot_observed_at":"2026-07-12T11:31:14.532101Z","title":"Timechat: A time-sensitive multimodal large language model for long video understanding, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.02551","last_updated":"2026-06-26T16:05:57Z","snapshot_observed_at":"2026-08-06T08:59:01.715141Z","submitted_at":"2026-06-26T16:05:57Z","title":"DELTAVID: Enhancing Fine-Grained Spatiotemporal Perception with Cross-Video Differences","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-07-12T11:31:14.532101Z"},"links":{"cited_paper":"/paper/2312.02051","citing_paper":"/paper/2607.02551"},"observation_digest":"sha256:39930770a98982e357e887112f914e18e3f46f238037e4186d04b901cc56a8ed","observation_id":"0e22faf1-5308-4869-a1b5-4ca00032fe8c","resolution":{"observed_at":"2026-07-12T11:31:14.532101Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.02051","last_updated":"2024-03-28T12:41:14Z","snapshot_observed_at":"2026-07-06T16:56:42.232555Z","submitted_at":"2023-12-04T17:09:52Z","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.02051","snapshot_observed_at":"2026-07-13T06:21:17.706173Z","title":"Timechat: A time-sensitive multi- modal large language model for long video understanding.arXiv preprint arXiv:2312.02051,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.08839","last_updated":"2026-07-09T18:00:47Z","snapshot_observed_at":"2026-07-15T23:17:54.548359Z","submitted_at":"2026-07-09T18:00:47Z","title":"Mixture of Probes: Learning from Privileged Modalities in Multimodal LLMs Through Probing","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-07-13T06:21:17.706173Z"},"links":{"cited_paper":"/paper/2312.02051","citing_paper":"/paper/2607.08839"},"observation_digest":"sha256:12509027b3252319a96fa41ccd33eafaa3bfb3f238e3de160b4d2191208ba709","observation_id":"ddf24105-f1b7-4d79-bbad-836acba77bba","resolution":{"observed_at":"2026-07-13T06:21:17.706173Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.02051","last_updated":"2024-03-28T12:41:14Z","snapshot_observed_at":"2026-07-06T16:56:42.232555Z","submitted_at":"2023-12-04T17:09:52Z","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.02051","snapshot_observed_at":"2026-08-01T18:04:15.439960Z","title":"Timechat: A time-sensitive multi- modal large language model for long video understanding.ArXiv, abs/2312.02051, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.17423","last_updated":"2026-07-19T22:04:33Z","snapshot_observed_at":"2026-08-05T19:15:04.907010Z","submitted_at":"2026-07-19T22:04:33Z","title":"TimeLens2: Generalist Video Temporal Grounding with Multimodal LLMs","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-01T18:04:15.439960Z"},"links":{"cited_paper":"/paper/2312.02051","citing_paper":"/paper/2607.17423"},"observation_digest":"sha256:75e928cf56d433c8269148004eb1e82392f4afa7dc73390a714b399f8356cbc4","observation_id":"b0c52bc0-b812-4f45-9d1b-c2b9ab76835b","resolution":{"observed_at":"2026-08-01T18:04:15.439960Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.02051","last_updated":"2024-03-28T12:41:14Z","snapshot_observed_at":"2026-07-06T16:56:42.232555Z","submitted_at":"2023-12-04T17:09:52Z","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.02051","snapshot_observed_at":"2026-07-31T23:32:53.515428Z","title":"Timechat: A time-sensitive multimodal large language model for long video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.23951","last_updated":"2026-07-27T02:56:43Z","snapshot_observed_at":"2026-07-31T23:32:36.040113Z","submitted_at":"2026-07-27T02:56:43Z","title":"TimePLE: Rethinking Temporal Representation for Video Temporal Grounding","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-07-31T23:32:53.515428Z"},"links":{"cited_paper":"/paper/2312.02051","citing_paper":"/paper/2607.23951"},"observation_digest":"sha256:15a9e1c3559f9033cdb90fcb91321fdf52056f3946523ea83761394e3330a4ef","observation_id":"75c7b714-5509-4238-bc47-182e8a457c22","resolution":{"observed_at":"2026-07-31T23:32:53.515428Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.02051","last_updated":"2024-03-28T12:41:14Z","snapshot_observed_at":"2026-07-06T16:56:42.232555Z","submitted_at":"2023-12-04T17:09:52Z","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.02051","snapshot_observed_at":"2026-08-04T17:24:16.109594Z","title":"2312.02051 , archivePrefix =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.01980","last_updated":"2026-08-03T09:42:42Z","snapshot_observed_at":"2026-08-08T12:51:52.547346Z","submitted_at":"2026-08-03T09:42:42Z","title":"AdaThinkV: Adaptive Thinking for Token-Efficient Video Reasoning","version":1},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-08-04T17:24:16.109594Z"},"links":{"cited_paper":"/paper/2312.02051","citing_paper":"/paper/2608.01980"},"observation_digest":"sha256:0b7f6f6465e36a4f923724c5f9ccbbd84572d4a677a58c70d23c3667fdb10982","observation_id":"7373f027-c75e-455f-a08b-410c4054f616","resolution":{"observed_at":"2026-08-04T17:24:16.109594Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2312.02051/citation-record","integrity":"/paper/2312.02051/integrity","json":"/paper/2312.02051/citation-record.json","paper":"/paper/2312.02051"},"outbound":[],"paper":{"arxiv_id":"2312.02051","last_updated":"2024-03-28T12:41:14Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T16:56:42.232555Z","submitted_at":"2023-12-04T17:09:52Z","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 27 inbound Pith citation observations for arXiv:2312.02051."}