{"as_of":"2026-08-11T09:08:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b22958ff2a39e71c1fdff92c1b7665413e9a02c7866f6cf6b0140a05286baa44","coverage":[{"denominator":59,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":59,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-13T23:30:00.457974Z","state":"measured"},{"denominator":159,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":159,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":209,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T05:41:47.548723Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":90,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2305.03726","last_updated":"2025-07-28T05:33:36Z","snapshot_observed_at":"2026-07-06T15:23:52.761222Z","submitted_at":"2023-05-05T17:59:46Z","title":"Otter: A Multi-Modal Model with In-Context Instruction Tuning","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-15T02:43:47.775691Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2305.03726"},"observation_digest":"sha256:c315bc864a256022decc3f3c97bf160a9d8de0d2573c3ddcc388ea08924282c9","observation_id":"ae4c2164-6c10-4cf3-84a5-8080ac845dcc","resolution":{"observed_at":"2026-05-15T02:43:47.862703Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2306.13549","last_updated":"2024-11-29T15:51:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-23T15:21:52Z","title":"A Survey on Multimodal Large Language Models","version":4},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-16T02:56:41.658658Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2306.13549"},"observation_digest":"sha256:3322cac590ae0bef68e08bc2971f09239c8ed6874272d3b9712ce910a45298c2","observation_id":"15850b87-1bb4-4c0d-8e85-e048d6ba2f49","resolution":{"observed_at":"2026-05-16T02:56:42.342701Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2307.06435","last_updated":"2024-10-17T01:10:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-12T20:01:52Z","title":"A Comprehensive Overview of Large Language Models","version":10},"reference_index":272,"source":"pdf_text","source_observed_at":"2026-05-19T20:28:38.900026Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2307.06435"},"observation_digest":"sha256:3d55c1979824fb41a7a1d5f8b16d01116a4659dc95526a2bffbbe43c2caba528","observation_id":"a365cf16-1b32-4160-b406-b9cbe99f8809","resolution":{"observed_at":"2026-05-19T20:28:39.220206Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2307.06942","last_updated":"2024-01-04T05:00:34Z","snapshot_observed_at":"2026-07-06T15:53:46.393481Z","submitted_at":"2023-07-13T17:58:32Z","title":"InternVid: A Large-scale Video-Text Dataset for Multimodal Understanding and Generation","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-15T06:30:22.431538Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2307.06942"},"observation_digest":"sha256:121f03635b774379e7d77ab25b010e96e5c36025c8426dde0e83935c869260e2","observation_id":"14ee9171-1cfa-45b2-95a4-f627cf2fd942","resolution":{"observed_at":"2026-05-15T06:30:22.509068Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2307.16125","last_updated":"2023-08-02T08:02:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-30T04:25:16Z","title":"SEED-Bench: Benchmarking Multimodal LLMs with Generative Comprehension","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-12T16:59:50.495335Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2307.16125"},"observation_digest":"sha256:f57d7d276e1777c177c14fc065164c93f8f998a0ff9669df60632c707d85f9c0","observation_id":"6abe26b8-437b-4edb-af3c-8c6a28a9c8bc","resolution":{"observed_at":"2026-05-13T23:30:00.879549Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2309.05922","last_updated":"2023-09-12T02:34:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-09-12T02:34:06Z","title":"A Survey of Hallucination in Large Foundation Models","version":1},"reference_index":130,"source":"arxiv_source","source_observed_at":"2026-05-16T15:21:00.778049Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2309.05922"},"observation_digest":"sha256:b32b65b94f442a6f7112a498249bc7fbc6e9b49418bf7083c9c9c74e1aceef40","observation_id":"a811ed80-2b38-4a2e-9e15-e364ded8ccdb","resolution":{"observed_at":"2026-05-16T15:21:00.932415Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2311.04257","last_updated":"2023-11-09T01:56:51Z","snapshot_observed_at":"2026-08-06T04:08:15.266897Z","submitted_at":"2023-11-07T14:21:29Z","title":"mPLUG-Owl2: Revolutionizing Multi-modal Large Language Model with Modality Collaboration","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-18T03:18:51.582340Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2311.04257"},"observation_digest":"sha256:646d126a9c4f958801afcddad370c40bedbf445b7b337ac64d5a051db15249f0","observation_id":"48b1f484-3639-4ead-a4b8-01f3bb01bddc","resolution":{"observed_at":"2026-05-18T03:18:51.688069Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2311.10122","last_updated":"2024-10-01T12:07:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-16T10:59:44Z","title":"Video-LLaVA: Learning United Visual Representation by Alignment Before Projection","version":3},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-05-14T18:08:01.166072Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2311.10122"},"observation_digest":"sha256:14f158e4caddcf33bce7ee521e82e5feaf7e1608b9fd2500ab06fa39c46d58f7","observation_id":"59472c27-18c4-4b95-b968-f6b1b8535212","resolution":{"observed_at":"2026-05-14T18:08:01.350653Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:3166837eefd6a7b730dd75d3619950b28683abb5cc4e38c53c8d3598540d22ea","observation_id":"884e9606-2235-40c7-b760-7076483c38d2","resolution":{"observed_at":"2026-05-17T20:22:35.099555Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2312.14238","last_updated":"2024-01-15T15:23:55Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-21T18:59:31Z","title":"InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks","version":3},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-05-13T22:46:09.693156Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2312.14238"},"observation_digest":"sha256:a98411188cad415e822e962fd5c7b893f17496859088eb102038a808ad2cf1e9","observation_id":"42f7cdf9-0992-4f98-b24e-edcbcea51670","resolution":{"observed_at":"2026-05-13T23:30:00.879549Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2401.03568","last_updated":"2024-01-25T21:20:27Z","snapshot_observed_at":"2026-07-30T06:39:40.880794Z","submitted_at":"2024-01-07T19:11:18Z","title":"Agent AI: Surveying the Horizons of Multimodal Interaction","version":2},"reference_index":84,"source":"arxiv_source","source_observed_at":"2026-05-18T14:25:58.876978Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2401.03568"},"observation_digest":"sha256:673d2f7cd03239ae87be43f38e9eb9624da6e75ff6355797fc463e9af7b6a9c3","observation_id":"239ac708-0def-4865-a4f0-ecb9cda9ba29","resolution":{"observed_at":"2026-05-18T14:25:59.287768Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2403.00476","last_updated":"2024-06-03T04:13:39Z","snapshot_observed_at":"2026-08-04T21:17:37.211833Z","submitted_at":"2024-03-01T12:02:19Z","title":"TempCompass: Do Video LLMs Really Understand Videos?","version":3},"reference_index":96,"source":"arxiv_source","source_observed_at":"2026-05-17T02:46:16.632743Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2403.00476"},"observation_digest":"sha256:78b345a5d48aa456d629e69c41afec3d5ebd269b229e2b662f9ff94aa2478a20","observation_id":"1b876105-a280-431c-ad71-0b69f8f9fd21","resolution":{"observed_at":"2026-05-17T02:46:16.711637Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2404.16821","last_updated":"2024-04-29T20:24:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-25T17:59:19Z","title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-12T20:58:58.849040Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2404.16821"},"observation_digest":"sha256:d7429b93569d4fcb5496706cb9008bb995d4ff3f97cbf55082e49ca4d8888ce2","observation_id":"a1e21a52-fc73-4886-9710-2f2d0bce3ea3","resolution":{"observed_at":"2026-05-13T23:30:00.879549Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2404.16994","last_updated":"2024-04-29T14:52:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-25T19:29:55Z","title":"PLLaVA : Parameter-free LLaVA Extension from Images to Videos for Video Dense Captioning","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-15T20:21:57.873354Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2404.16994"},"observation_digest":"sha256:6d69bd8dd4cb6de6b17bbda41e3af210bde9f123d017696618d912bfe4e4370e","observation_id":"fac87720-fb17-40cd-bd06-1a749f8a1017","resolution":{"observed_at":"2026-05-15T20:21:57.962597Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2406.04264","last_updated":"2025-01-01T15:53:58Z","snapshot_observed_at":"2026-08-03T20:38:36.602554Z","submitted_at":"2024-06-06T17:09:32Z","title":"MLVU: Benchmarking Multi-task Long Video Understanding","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-14T19:55:26.333923Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2406.04264"},"observation_digest":"sha256:e445be22501cef767b167ee645187ff4fab3e9d2c558ec552ea06a1a71a079c0","observation_id":"a73ef674-3145-4c45-a6a8-0b1191e21e94","resolution":{"observed_at":"2026-05-14T19:55:26.590413Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2406.05615","last_updated":"2026-05-12T03:37:41Z","snapshot_observed_at":"2026-07-06T18:27:36.903877Z","submitted_at":"2024-06-09T02:36:28Z","title":"Video-Language Understanding: A Survey from Model Architecture, Model Training, and Data Perspectives","version":4},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-24T00:22:35.635679Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2406.05615"},"observation_digest":"sha256:e391edbbd071dedb058bf90f3a2fbbf8f10c6629954aa7d94ecd300c8790929d","observation_id":"bd69b2c4-4d4e-44ed-91b8-551ffb3f7747","resolution":{"observed_at":"2026-05-24T00:23:39.617174Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2407.03320","last_updated":"2024-07-03T17:59:21Z","snapshot_observed_at":"2026-08-04T22:09:42.241578Z","submitted_at":"2024-07-03T17:59:21Z","title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-05-17T10:46:28.447347Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2407.03320"},"observation_digest":"sha256:24a5b0ec73dd22d6c3a0a988dcdc88d5a3cafa066bf59c112499b662ffdbb1f4","observation_id":"4015d4dd-a620-425c-920c-7210c1415dc2","resolution":{"observed_at":"2026-05-17T10:46:28.625942Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2407.07895","last_updated":"2024-07-28T19:58:08Z","snapshot_observed_at":"2026-07-06T18:44:24.873040Z","submitted_at":"2024-07-10T17:59:43Z","title":"LLaVA-NeXT-Interleave: Tackling Multi-image, Video, and 3D in Large Multimodal Models","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-11T06:01:53.730356Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2407.07895"},"observation_digest":"sha256:0da01f0973114548c99578a568202f21356a0f4eca30e64f88cd8e1043f0d561","observation_id":"00f85e25-659b-4493-a6d5-69c64df88929","resolution":{"observed_at":"2026-05-13T23:30:00.879549Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":222,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:4dba849d1dbc076f4fb9f9162d0f6eed5a3bd88be81d88e1eeebfff19227e059","observation_id":"a6cc0f65-2612-433c-aac3-aaeb76eeb852","resolution":{"observed_at":"2026-05-20T06:20:36.423910Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2408.16500","last_updated":"2024-08-29T12:59:12Z","snapshot_observed_at":"2026-08-11T07:53:33.804680Z","submitted_at":"2024-08-29T12:59:12Z","title":"CogVLM2: Visual Language Models for Image and Video Understanding","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-16T20:10:27.633010Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2408.16500"},"observation_digest":"sha256:edc496a1797c0cf8f797aa1170e61c3aa2bd89e8c3b6f858aaf7c7d638711212","observation_id":"0652cd27-6b42-445f-8066-cb5879896a28","resolution":{"observed_at":"2026-05-16T20:10:27.722354Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:f6a8de3bea839f69c24cd95b80a40f4d880848051f1db4657e86c7f49887040e","observation_id":"945090e3-98eb-48ff-a09f-1ec94a9f1394","resolution":{"observed_at":"2026-05-13T23:30:00.879549Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2411.16771","last_updated":"2026-04-23T13:21:19Z","snapshot_observed_at":"2026-08-02T03:20:43.405143Z","submitted_at":"2024-11-25T06:17:23Z","title":"VidHal: Benchmarking Temporal Hallucinations in Vision LLMs","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-23T16:57:12.821916Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2411.16771"},"observation_digest":"sha256:295dd55aee35edcd32ad0303fd7594daa3a88c95af9b152b680a769cd8d8e1c0","observation_id":"79b5e6c1-457f-4a0e-a0ab-f3722dbca13d","resolution":{"observed_at":"2026-05-23T16:58:12.013422Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2412.02930","last_updated":"2026-04-19T04:24:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-04T00:50:33Z","title":"TemporalVLM: Video LLMs for Temporal Reasoning in Long Videos","version":6},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-23T08:08:01.889675Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2412.02930"},"observation_digest":"sha256:644ffc8356a97741def47be0ff86b7b898ba3f925b061dc361b9dc6096888044","observation_id":"ac1221c8-b61b-4081-8dd9-81e78e4a7685","resolution":{"observed_at":"2026-05-23T08:12:43.943872Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2412.05271","last_updated":"2025-09-26T12:52:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-06T18:57:08Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","version":5},"reference_index":130,"source":"pdf_text","source_observed_at":"2026-05-10T13:23:57.588851Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2412.05271"},"observation_digest":"sha256:9b391d8b09cee97d635cf2e3d6e06dc7517b289f48bc21ced9642812df734e4c","observation_id":"56d1e175-a9ee-43b8-95c6-8e79068f73e4","resolution":{"observed_at":"2026-05-13T23:30:00.879549Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2412.06224","last_updated":"2025-02-06T10:14:36Z","snapshot_observed_at":"2026-08-06T17:32:08.916929Z","submitted_at":"2024-12-09T05:55:55Z","title":"Uni-NaVid: A Video-based Vision-Language-Action Model for Unifying Embodied Navigation Tasks","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-16T19:51:36.137985Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2412.06224"},"observation_digest":"sha256:0e89f00e080a1455cfa28fa87563db67e937566deed56fe257c7d520f0b3072a","observation_id":"419570b1-1595-48dd-9558-1b75b0c809b7","resolution":{"observed_at":"2026-05-16T19:51:36.324238Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-11T05:41:47.548723Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.17295","last_updated":"2024-12-23T05:32:48Z","snapshot_observed_at":"2026-08-11T05:34:54.756252Z","submitted_at":"2024-12-23T05:32:48Z","title":"Friends-MMC: A Dataset for Multi-modal Multi-party Conversation Understanding","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-11T05:41:47.548723Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2412.17295"},"observation_digest":"sha256:189968f37fa7bdb5d95fe3626eec0dc29fbfb1030e15a82d482b58f4face8c60","observation_id":"729ca95e-ca18-4b1a-b029-2b6b82e69251","resolution":{"observed_at":"2026-08-11T05:41:47.548723Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2412.17574","last_updated":"2026-04-13T15:05:36Z","snapshot_observed_at":"2026-07-06T20:12:09.120832Z","submitted_at":"2024-12-23T13:45:56Z","title":"HumanVBench: Probing Human-Centric Video Understanding in MLLMs with Automatically Synthesized Benchmarks","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-23T07:05:08.716223Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2412.17574"},"observation_digest":"sha256:3cbed0296bce89c78be07f3361dc7cd2089057fe9bed925e756e2e2a72a568ef","observation_id":"8d0bae63-f0bf-4c94-845d-8e3eb0568304","resolution":{"observed_at":"2026-05-23T07:05:29.173667Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-11T00:47:12.242711Z","title":"Videochat: Chat-centric video understanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.19326","last_updated":"2025-06-30T13:15:13Z","snapshot_observed_at":"2026-08-11T00:39:56.306346Z","submitted_at":"2024-12-26T18:56:05Z","title":"Task Preference Optimization: Improving Multimodal Large Language Models with Vision Task Alignment","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-11T00:47:12.242711Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2412.19326"},"observation_digest":"sha256:8491d49cc71039df40f4bbf5a66f28eac03d00c226e047de28e1387bfcbd006b","observation_id":"bc44888b-8097-4936-ade0-db1a18d98b24","resolution":{"observed_at":"2026-08-11T00:47:12.242711Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-10T23:07:14.533205Z","title":"Videochat: Chat-centric video understanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.21080","last_updated":"2024-12-30T16:57:05Z","snapshot_observed_at":"2026-08-10T23:00:36.632749Z","submitted_at":"2024-12-30T16:57:05Z","title":"Vinci: A Real-time Embodied Smart Assistant based on Egocentric Vision-Language Model","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-10T23:07:14.533205Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2412.21080"},"observation_digest":"sha256:2586c2cf4432761a0edf42e23f1642da7d1ebdad2ef4706ea223b6957aabdcb3","observation_id":"ab47944e-5706-4a95-8f8d-40b98bedb0f2","resolution":{"observed_at":"2026-08-10T23:07:14.533205Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-10T22:52:00.849222Z","title":"Videochat: Chat-centric video understanding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00513","last_updated":"2025-03-18T16:01:24Z","snapshot_observed_at":"2026-08-10T22:46:57.431139Z","submitted_at":"2024-12-31T15:53:50Z","title":"CaReBench: A Fine-Grained Benchmark for Video Captioning and Retrieval","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:00.849222Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2501.00513"},"observation_digest":"sha256:0bcca3abade478e0c85b2d65ce32a3c46b71cce6052352813662b4940f1d0b60","observation_id":"349855bf-91db-4b2c-b76a-3e2ed6b978dd","resolution":{"observed_at":"2026-08-10T22:52:00.849222Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2501.00574","last_updated":"2025-07-13T16:21:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-31T18:01:23Z","title":"VideoChat-Flash: Hierarchical Compression for Long-Context Video Modeling","version":4},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-18T04:02:43.261543Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2501.00574"},"observation_digest":"sha256:4dc4b92202502bf65a568f583fd6776b96de2f97216ee67df21a632a36d612a0","observation_id":"7c3bf99c-f3e6-405f-a78d-f220407e7788","resolution":{"observed_at":"2026-05-18T04:02:43.438898Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-10T22:57:40.045508Z","title":"Videochat: Chat-centric video understanding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00584","last_updated":"2025-04-17T10:10:16Z","snapshot_observed_at":"2026-08-11T01:33:05.827241Z","submitted_at":"2024-12-31T18:17:05Z","title":"Online Video Understanding: OVBench and VideoChat-Online","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-10T22:57:40.045508Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2501.00584"},"observation_digest":"sha256:b5cbdec5cf3b52bf1f192229d0a8c7ee37624d0ca80c860fb273203ce6c26038","observation_id":"225d188f-b04d-49e7-b30b-e3790e154e73","resolution":{"observed_at":"2026-08-10T22:57:40.045508Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-10T22:19:53.372394Z","title":"Videochat: Chat-centric video understanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.02135","last_updated":"2025-01-03T23:03:24Z","snapshot_observed_at":"2026-08-10T23:06:48.073542Z","submitted_at":"2025-01-03T23:03:24Z","title":"AVTrustBench: Assessing and Enhancing Reliability and Robustness in Audio-Visual LLMs","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-10T22:19:53.372394Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2501.02135"},"observation_digest":"sha256:2a7124e0a63738d43537f9039400a66f56f3ae79f7a9777abc8ef84259457a9e","observation_id":"5ce8e1bb-2c42-4e4a-9e91-a9918787bddf","resolution":{"observed_at":"2026-08-10T22:19:53.372394Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-10T22:08:09.456281Z","title":"Videochat: Chat-centric video understanding,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.02765","last_updated":"2025-01-06T05:15:59Z","snapshot_observed_at":"2026-08-11T04:56:13.579583Z","submitted_at":"2025-01-06T05:15:59Z","title":"Visual Large Language Models for Generalized and Specialized Applications","version":1},"reference_index":146,"source":"pdf_text","source_observed_at":"2026-08-10T22:08:09.456281Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2501.02765"},"observation_digest":"sha256:8b2ccaa23f75ce07688e85921d4b81a543ce72d36cb9b86b4733f5b7243f82d9","observation_id":"f9447c53-292f-4a83-a700-71de2682f56d","resolution":{"observed_at":"2026-08-10T22:08:09.456281Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2501.02955","last_updated":"2026-05-12T15:02:48Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-06T11:57:38Z","title":"MotionBench: Benchmarking and Improving Fine-grained Video Motion Understanding for Vision Language Models","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-23T05:44:31.546843Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2501.02955"},"observation_digest":"sha256:6efafbee88a9dc12ae02aebedb3e5ea004cf6b954f7fba9f254e1efdb9e6816e","observation_id":"52152dbd-8747-428f-ad2a-f626394b2450","resolution":{"observed_at":"2026-05-23T05:45:28.365231Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2501.04001","last_updated":"2025-11-03T17:35:29Z","snapshot_observed_at":"2026-08-08T01:58:42.644918Z","submitted_at":"2025-01-07T18:58:54Z","title":"Sa2VA: Marrying SAM2 with LLaVA for Dense Grounded Understanding of Images and Videos","version":3},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-16T11:39:22.340737Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2501.04001"},"observation_digest":"sha256:0e8e6706695b1b9067809c4781caa841be6dfe779bd52ad8b6fa1d66e19ef74c","observation_id":"0218fc29-b73d-4169-9e62-bce6235bc050","resolution":{"observed_at":"2026-05-16T11:39:22.552419Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-10T21:42:11.565112Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.04302","last_updated":"2025-01-08T06:26:16Z","snapshot_observed_at":"2026-08-10T21:34:19.062468Z","submitted_at":"2025-01-08T06:26:16Z","title":"H-MBA: Hierarchical MamBa Adaptation for Multi-Modal Video Understanding in Autonomous Driving","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-10T21:42:11.565112Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2501.04302"},"observation_digest":"sha256:97103919895c91567254a50274e58a9fc51a7dac8cb21e90a14da5fbe0343625","observation_id":"725677b8-7fb1-41e6-9608-fd9506f47da7","resolution":{"observed_at":"2026-08-10T21:42:11.565112Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-10T21:23:57.923377Z","title":"Videochat: Chat-centric video understanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.05037","last_updated":"2025-03-27T09:39:11Z","snapshot_observed_at":"2026-08-10T21:17:52.245219Z","submitted_at":"2025-01-09T07:51:14Z","title":"LongViTU: Instruction Tuning for Long-Form Video Understanding","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-10T21:23:57.923377Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2501.05037"},"observation_digest":"sha256:7b3b2d77721046143bf0d60663c684bda03934b6ae70ee7f3306fbb18e5f6c8e","observation_id":"1730d6a1-7a2d-4750-9cbe-f79c1e5e14d0","resolution":{"observed_at":"2026-08-10T21:23:57.923377Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2501.05067","last_updated":"2026-04-20T07:42:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-09T08:43:57Z","title":"LLaVA-Octopus: Unlocking Instruction-Driven Adaptive Projector Fusion for Video Understanding","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-23T06:01:00.775721Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2501.05067"},"observation_digest":"sha256:9f214c046c19e2edfa738571faf66a35fc565a553649b4b8ef49ae582e74c979","observation_id":"08b776a3-e1f5-41b6-88f4-15a3c3c6c2a2","resolution":{"observed_at":"2026-05-23T06:02:37.497349Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-10T20:54:13.259565Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.06761","last_updated":"2025-01-12T10:08:26Z","snapshot_observed_at":"2026-08-10T20:47:56.124760Z","submitted_at":"2025-01-12T10:08:26Z","title":"VidChain: Chain-of-Tasks with Metric-based Direct Preference Optimization for Dense Video Captioning","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-10T20:54:13.259565Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2501.06761"},"observation_digest":"sha256:1f9f6ee4ae3844c857808f5c6a9a82255efdceb472195674a6a3abab9da8c0d4","observation_id":"2c5330c3-8e67-4e3d-a3e0-c4252774ad4f","resolution":{"observed_at":"2026-08-10T20:54:13.259565Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-10T20:33:54.040489Z","title":"Videochat: Chat-centric video understanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07978","last_updated":"2025-01-14T09:52:56Z","snapshot_observed_at":"2026-08-11T08:54:41.859500Z","submitted_at":"2025-01-14T09:52:56Z","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:54.040489Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2501.07978"},"observation_digest":"sha256:ab9b42b0ffd0a3855340732a5c64c50c5103ac96a2bc69e0fbb77d6104c5a991","observation_id":"2e33504e-cdb5-4ea3-9099-cc2a3bfd6e84","resolution":{"observed_at":"2026-08-10T20:33:54.040489Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-10T20:34:11.725520Z","title":"Videochat: Chat-centric video understanding,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.08168","last_updated":"2025-01-14T14:49:45Z","snapshot_observed_at":"2026-08-10T20:27:11.223403Z","submitted_at":"2025-01-14T14:49:45Z","title":"LeapVAD: A Leap in Autonomous Driving via Cognitive Perception and Dual-Process Thinking","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:11.725520Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2501.08168"},"observation_digest":"sha256:329be0a97f6a07facfcbc2601e235c83b03d368e19a2166ec8ebcd576df6e36b","observation_id":"567c6c6b-9aa7-4829-8019-bd2f95cc1b56","resolution":{"observed_at":"2026-08-10T20:34:11.725520Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-10T20:03:12.859584Z","title":"Videochat: Chat-centric video understanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.09502","last_updated":"2025-01-16T12:27:05Z","snapshot_observed_at":"2026-08-10T19:55:29.245089Z","submitted_at":"2025-01-16T12:27:05Z","title":"Omni-Emotion: Extending Video MLLM with Detailed Face and Audio Modeling for Multimodal Emotion Analysis","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-10T20:03:12.859584Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2501.09502"},"observation_digest":"sha256:6fa26aeb7870cc86bde737d39fabd547bfe11c1251df599eaabe171a591af869","observation_id":"6cbe9747-b6f5-453c-befc-6c2420fc9e32","resolution":{"observed_at":"2026-08-10T20:03:12.859584Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2501.13106","last_updated":"2025-06-03T03:33:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T18:59:46Z","title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","version":4},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-11T01:19:59.603343Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2501.13106"},"observation_digest":"sha256:8644d480669502c59a2545a889c84ed04658b96efd2c71118a7a8ef18369fc8e","observation_id":"34fe1b59-3cca-4a37-b96a-1ea327675c95","resolution":{"observed_at":"2026-05-13T23:30:00.879549Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-10T15:56:37.565527Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13536","last_updated":"2025-01-23T10:35:22Z","snapshot_observed_at":"2026-08-10T15:49:26.456054Z","submitted_at":"2025-01-23T10:35:22Z","title":"ReasVQA: Advancing VideoQA with Imperfect Reasoning Process","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-10T15:56:37.565527Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2501.13536"},"observation_digest":"sha256:cdb97a72624d97889a5032ad1d5bda0c05ad2ca058a8c6e54a35b3c7cd95906a","observation_id":"74613560-8f9d-42b4-8ac0-78b3b5004867","resolution":{"observed_at":"2026-08-10T15:56:37.565527Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-10T15:35:30.172948Z","title":"Videochat: Chat-centric video understanding.arXiv preprint arXiv:2305.06355, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13919","last_updated":"2025-09-01T07:50:58Z","snapshot_observed_at":"2026-08-11T01:32:58.320917Z","submitted_at":"2025-01-23T18:58:03Z","title":"Temporal Preference Optimization for Long-Form Video Understanding","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-10T15:35:30.172948Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2501.13919"},"observation_digest":"sha256:7f9ac1572ddfea9585abf69db097544fe47a3524eff4982919ef84fd4c3aa8ae","observation_id":"a4755349-b04c-4fda-8bd6-7c526c9e1545","resolution":{"observed_at":"2026-08-10T15:35:30.172948Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-10T14:40:55.240225Z","title":"Videochat: Chat-centric video understanding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.15111","last_updated":"2025-01-25T07:26:37Z","snapshot_observed_at":"2026-08-10T14:34:53.812449Z","submitted_at":"2025-01-25T07:26:37Z","title":"HumanOmni: A Large Vision-Speech Language Model for Human-Centric Video Understanding","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-10T14:40:55.240225Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2501.15111"},"observation_digest":"sha256:4b8602671394e1c3f8e40686c07fcd858716b22d39bcd12a114e8a4636ae8b67","observation_id":"6b1dd5db-994e-46ee-a79a-44ee7d9636a0","resolution":{"observed_at":"2026-08-10T14:40:55.240225Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-10T14:18:17.605226Z","title":"Videochat: Chat-centric video understanding.arXiv preprint arXiv:2305.06355, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.15513","last_updated":"2025-06-10T14:30:19Z","snapshot_observed_at":"2026-08-10T14:10:29.461568Z","submitted_at":"2025-01-26T13:10:12Z","title":"TinyLLaVA-Video: Towards Smaller LMMs for Video Understanding with Group Resampler","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-10T14:18:17.605226Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2501.15513"},"observation_digest":"sha256:cd488ad4f83d910ce109fb688039eccf3fd816b08cc4cbb2a5b9149b5e673983","observation_id":"39898d34-2355-4ed0-82d8-51ca383af8ed","resolution":{"observed_at":"2026-08-10T14:18:17.605226Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-10T13:53:26.184775Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.15953","last_updated":"2025-01-27T10:57:24Z","snapshot_observed_at":"2026-08-10T13:47:43.852671Z","submitted_at":"2025-01-27T10:57:24Z","title":"Understanding Long Videos via LLM-Powered Entity Relation Graphs","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-10T13:53:26.184775Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2501.15953"},"observation_digest":"sha256:d445178da745dce4e1e6498ef05c919c2c7277788edcd48795216a201abd8fd2","observation_id":"7813284f-12ab-4ccb-bca4-624e0e947a5e","resolution":{"observed_at":"2026-08-10T13:53:26.184775Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-09T21:21:45.654533Z","title":"Videochat: Chat-centric video understanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.19098","last_updated":"2025-05-19T10:18:07Z","snapshot_observed_at":"2026-08-10T11:57:26.754523Z","submitted_at":"2025-01-31T12:45:46Z","title":"$\\infty$-Video: A Training-Free Approach to Long Video Understanding via Continuous-Time Memory Consolidation","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-09T21:21:45.654533Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2501.19098"},"observation_digest":"sha256:9986ef3c836abc2ee06f99fd622afc5e3283930fde4b16e05533b7d80abc1bd2","observation_id":"a29aa8fa-f608-47fe-bd84-ffa7172425ed","resolution":{"observed_at":"2026-08-09T21:21:45.654533Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-09T15:04:40.281075Z","title":"Videochat: Chat-centric video understanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.01524","last_updated":"2025-02-03T17:01:59Z","snapshot_observed_at":"2026-08-09T21:17:23.790261Z","submitted_at":"2025-02-03T17:01:59Z","title":"Efficiently Integrate Large Language Models with Visual Perception: A Survey from the Training Paradigm Perspective","version":1},"reference_index":139,"source":"pdf_text","source_observed_at":"2026-08-09T15:04:40.281075Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2502.01524"},"observation_digest":"sha256:db1caf1713bbca4257d2de90af92ca9a233f17a7fc49a4de3ca80f517199f185","observation_id":"58b706f8-00fe-41a4-8b9f-decae703afd1","resolution":{"observed_at":"2026-08-09T15:04:40.281075Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-08T21:12:22.802431Z","title":"Videochat: Chat-centric video understanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.05240","last_updated":"2025-02-12T14:43:02Z","snapshot_observed_at":"2026-08-10T00:38:45.519768Z","submitted_at":"2025-02-07T12:18:20Z","title":"Survey on AI-Generated Media Detection: From Non-MLLM to MLLM","version":2},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-08T21:12:22.802431Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2502.05240"},"observation_digest":"sha256:8381ed59aecc44dfb7d8f4e668c05be3e89df27e899aac5b7e856f605d461c07","observation_id":"9573671c-4511-4233-b5bb-cc2db817550c","resolution":{"observed_at":"2026-08-08T21:12:22.802431Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2503.12605","last_updated":"2025-03-23T13:47:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-16T18:39:13Z","title":"Multimodal Chain-of-Thought Reasoning: A Comprehensive Survey","version":2},"reference_index":198,"source":"pdf_text","source_observed_at":"2026-05-15T17:18:52.996467Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2503.12605"},"observation_digest":"sha256:186f10c78ae5aca5f380cb3ebfda27e3eab6c29bf70e04007c382a434c63622b","observation_id":"7ff34654-d8bb-444b-98b8-ecc56fb6f257","resolution":{"observed_at":"2026-05-15T17:18:53.359724Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2503.14505","last_updated":"2026-05-03T19:49:25Z","snapshot_observed_at":"2026-08-06T04:26:20.176735Z","submitted_at":"2025-03-18T17:59:58Z","title":"MusicInfuser: Making Video Diffusion Listen and Dance","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-22T23:28:51.767845Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2503.14505"},"observation_digest":"sha256:556e227c25d1a85b8e8ef6b5bd2f675150cd179f74226c1a61693f8719c30978","observation_id":"61c4b2fd-55ba-44d4-aff9-e9fe527c97fd","resolution":{"observed_at":"2026-05-22T23:32:15.645719Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2504.06958","last_updated":"2025-11-11T08:30:00Z","snapshot_observed_at":"2026-08-02T02:31:33.589341Z","submitted_at":"2025-04-09T15:09:27Z","title":"VideoChat-R1: Enhancing Spatio-Temporal Perception via Reinforcement Fine-Tuning","version":5},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-15T20:56:07.247122Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2504.06958"},"observation_digest":"sha256:3f25c834a8f1e3aaf3958b40bafd0b7cdc55f73d2483bf661fecf1bd651557fc","observation_id":"660889a1-6a46-4982-a916-8f1ef2e247ea","resolution":{"observed_at":"2026-05-15T20:56:07.809487Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2504.10479","last_updated":"2025-04-19T03:47:21Z","snapshot_observed_at":"2026-08-10T18:37:57.419939Z","submitted_at":"2025-04-14T17:59:25Z","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","version":3},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-10T13:41:07.991012Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2504.10479"},"observation_digest":"sha256:2e81537aab0663f0059feeb513369b8ec6badaa46e98620a420c6e4d2db2c894","observation_id":"96dde0f3-38d0-406b-b44d-5595cf56c437","resolution":{"observed_at":"2026-05-13T23:30:00.879549Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T15:34:42.546538Z","title":"Videochat: Chat-centric video understanding.arXiv preprint arXiv:2305.06355, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-10T08:26:12.719387Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:42.546538Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:f107fe1e0afd4cf5950ced246e5c1bb72bb664374d98ce66ba813ffd5c4936f7","observation_id":"3f9b5974-c0ef-464f-93fd-7d8253ab8b18","resolution":{"observed_at":"2026-08-07T15:34:42.546538Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T15:19:12.678804Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17114","last_updated":"2025-09-05T17:40:14Z","snapshot_observed_at":"2026-08-10T17:14:45.967006Z","submitted_at":"2025-05-21T14:33:36Z","title":"RAVEN: Query-Guided Representation Alignment for Question Answering over Audio, Video, Embedded Sensors, and Natural Language","version":3},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-07T15:19:12.678804Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2505.17114"},"observation_digest":"sha256:5d6c2caa028310e3f23dd6f3fe3d5bd88c1cb6eddffd4d4d89a8d2542a69c063","observation_id":"98ed9a7a-80ee-42db-9d19-2b2be5e63aed","resolution":{"observed_at":"2026-08-07T15:19:12.678804Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T14:37:49.563235Z","title":"Videochat: Chat-centric video understanding.arXiv preprint arXiv:2305.06355, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.18115","last_updated":"2025-05-23T17:14:12Z","snapshot_observed_at":"2026-08-10T17:27:09.159914Z","submitted_at":"2025-05-23T17:14:12Z","title":"Instructify: Demystifying Metadata to Visual Instruction Tuning Data Conversion","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T14:37:49.563235Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2505.18115"},"observation_digest":"sha256:02abb9512f2150b6929224d5599c9a818de6efd914863c8ce79afdb5f2cadf73","observation_id":"faaad85f-8620-4213-b570-1c8ea8c4ffc9","resolution":{"observed_at":"2026-08-07T14:37:49.563235Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T14:24:08.311582Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19125","last_updated":"2025-05-25T12:44:12Z","snapshot_observed_at":"2026-08-10T12:03:06.094107Z","submitted_at":"2025-05-25T12:44:12Z","title":"RTime-QA: A Benchmark for Atomic Temporal Event Understanding in Large Multi-modal Models","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:08.311582Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2505.19125"},"observation_digest":"sha256:ad89272639627178356e5fb9ebd679c3d91a9d74cd3eaa23a16c6a3032ee25fb","observation_id":"8de95b1e-1fd5-40ea-b53d-1fe02be8f7fa","resolution":{"observed_at":"2026-08-07T14:24:08.311582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T14:18:05.900446Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19554","last_updated":"2025-05-26T06:17:21Z","snapshot_observed_at":"2026-08-09T08:50:26.066997Z","submitted_at":"2025-05-26T06:17:21Z","title":"Aggregated Structural Representation with Large Language Models for Human-Centric Layout Generation","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-07T14:18:05.900446Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2505.19554"},"observation_digest":"sha256:4fa999b9a47b383bad671480ae6f5ec6873277cd953bb1ac40bfa77d8f811088","observation_id":"982636d1-66c4-4956-ba0f-a16b661028ac","resolution":{"observed_at":"2026-08-07T14:18:05.900446Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T14:09:08.616938Z","title":"Videochat: Chat-centric video understanding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.19877","last_updated":"2025-05-26T12:05:16Z","snapshot_observed_at":"2026-08-10T09:10:24.017088Z","submitted_at":"2025-05-26T12:05:16Z","title":"Vad-R1: Towards Video Anomaly Reasoning via Perception-to-Cognition Chain-of-Thought","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T14:09:08.616938Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2505.19877"},"observation_digest":"sha256:56d8c1e3d255369b68a8aa96d1a9db00da34f18765e42a287109c7f48aaac1bf","observation_id":"055477c3-d512-49c2-84aa-4834b945b15b","resolution":{"observed_at":"2026-08-07T14:09:08.616938Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T13:48:13.144440Z","title":"Videochat: Chat-centric video understanding.arXiv preprint arXiv:2305.06355, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.20920","last_updated":"2025-05-27T09:10:59Z","snapshot_observed_at":"2026-08-09T07:43:10.959148Z","submitted_at":"2025-05-27T09:10:59Z","title":"HuMoCon: Concept Discovery for Human Motion Understanding","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T13:48:13.144440Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2505.20920"},"observation_digest":"sha256:2cad820fa426d56c55974f0334d827d4dd37a4bca9210d21e3ddf928b2ebf47a","observation_id":"57916962-adad-4b7f-9fd7-09b260c88146","resolution":{"observed_at":"2026-08-07T13:48:13.144440Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2505.21374","last_updated":"2025-05-27T16:05:01Z","snapshot_observed_at":"2026-08-06T04:32:21.352745Z","submitted_at":"2025-05-27T16:05:01Z","title":"Video-Holmes: Can MLLM Think Like Holmes for Complex Video Reasoning?","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-17T05:40:55.944288Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2505.21374"},"observation_digest":"sha256:06e394283a8a6f4aac80be02224acb698eb96cc5e1f07be5560164d3eb78bbe9","observation_id":"5bf2a2c0-36b0-4b24-abcb-c95a924b766b","resolution":{"observed_at":"2026-05-17T05:40:55.999728Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T13:35:52.839627Z","title":"Videochat: Chat-centric video understanding.arXiv preprint arXiv:2305.06355, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.21494","last_updated":"2025-05-27T17:56:57Z","snapshot_observed_at":"2026-08-07T21:17:47.870634Z","submitted_at":"2025-05-27T17:56:57Z","title":"Adversarial Attacks against Closed-Source MLLMs via Feature Optimal Alignment","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T13:35:52.839627Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2505.21494"},"observation_digest":"sha256:d216a0187c2df07e8892dfd6750544b3cf1b2becd883dbe8417ea287c28b3b58","observation_id":"23b3972f-bd30-42bf-b519-bc4af4101549","resolution":{"observed_at":"2026-08-07T13:35:52.839627Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T12:45:46.512570Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.23693","last_updated":"2025-05-29T17:31:13Z","snapshot_observed_at":"2026-08-07T18:41:28.769570Z","submitted_at":"2025-05-29T17:31:13Z","title":"VF-Eval: Evaluating Multimodal LLMs for Generating Feedback on AIGC Videos","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-07T12:45:46.512570Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2505.23693"},"observation_digest":"sha256:8c0fbfc7698e2f4712a1b3f8398b2379d9beee5ca356a392c8cfd1a5ba9fa4b0","observation_id":"cfb42be7-2600-43d1-bc3f-806577ce11ec","resolution":{"observed_at":"2026-08-07T12:45:46.512570Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2505.23747","last_updated":"2026-05-19T02:23:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-29T17:59:04Z","title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-16T08:34:36.824053Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2505.23747"},"observation_digest":"sha256:a735a78c2f5aae433a7df9690f78a28f6d38093d9a3f8ea43d1ced51fbabdd05","observation_id":"3a7c88d5-139f-40ae-9010-2a4162c88189","resolution":{"observed_at":"2026-05-16T08:34:36.917365Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":"2305.06355","doi":"10.48550/arxiv.2305.06355","metadata_source":"pith","pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VideoChat: Chat-Centric Video Understanding","venue":"cs.CV","work_id":"07461eec-156c-4054-a28e-b84bc53bf6e1","year":2023},"citing_paper":{"arxiv_id":"2505.23747","last_updated":"2026-05-19T02:23:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-29T17:59:04Z","title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-22T00:59:13.826054Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2505.23747"},"observation_digest":"sha256:00989a79dd2f52f9c6221244b9c515a8394fe1d36b091e0e256b82325f5787c5","observation_id":"e02ec366-64f2-4304-8fe3-353a6c33a29a","resolution":{"observed_at":"2026-05-22T01:00:51.411925Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T12:37:19.138328Z","title":"Videochat: Chat-centric video understanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:19.138328Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:b8728ecc471c52cc0193950b6d56fa7760a292308901431c3aa429af7a75e78b","observation_id":"60b66c9c-95ff-4c99-aa4f-806c7132fb93","resolution":{"observed_at":"2026-08-07T12:37:19.138328Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T12:32:51.944782Z","title":"Videochat: Chat-centric video understanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.24329","last_updated":"2025-07-31T03:03:18Z","snapshot_observed_at":"2026-08-09T06:10:22.489105Z","submitted_at":"2025-05-30T08:10:18Z","title":"DisTime: Distribution-based Time Representation for Video Large Language Models","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T12:32:51.944782Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2505.24329"},"observation_digest":"sha256:f8cce70d563619886e0e4d1f9d854b13b18a598c064c11733c1b6ee9b8e72d94","observation_id":"21d7e694-12db-4d82-a923-8fecfdde71d9","resolution":{"observed_at":"2026-08-07T12:32:51.944782Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T12:32:13.451229Z","title":"Videochat: Chat-centric video understanding.arXiv preprint arXiv:2305.06355, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.24346","last_updated":"2025-05-30T08:39:36Z","snapshot_observed_at":"2026-08-09T11:59:19.522063Z","submitted_at":"2025-05-30T08:39:36Z","title":"VUDG: A Dataset for Video Understanding Domain Generalization","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T12:32:13.451229Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2505.24346"},"observation_digest":"sha256:e3b8b331ed9dcd714c74d7f3a4a254bb7d08441b7374dfea5fbe5d354f5610a9","observation_id":"7ddc9ed9-fd2d-4530-ba5f-289c8eae72c8","resolution":{"observed_at":"2026-08-07T12:32:13.451229Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T12:30:31.859767Z","title":"VideoChat: Chat-centric video under- standing,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.24371","last_updated":"2025-07-28T02:22:21Z","snapshot_observed_at":"2026-08-10T08:13:09.334545Z","submitted_at":"2025-05-30T09:04:30Z","title":"Grid-LOGAT: Grid Based Local and Global Area Transcription for Video Question Answering","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T12:30:31.859767Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2505.24371"},"observation_digest":"sha256:19acf44c31c83c7d2f2616eda397dec9aae8b25676e02349cec3a541d4ac8c26","observation_id":"ac08881b-48a1-4659-a175-f4a53b64c0d8","resolution":{"observed_at":"2026-08-07T12:30:31.859767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T12:26:37.478169Z","title":"Videochat: Chat-centric video understanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.24476","last_updated":"2025-05-30T11:23:21Z","snapshot_observed_at":"2026-08-09T05:31:37.592410Z","submitted_at":"2025-05-30T11:23:21Z","title":"Period-LLM: Extending the Periodic Capability of Multimodal Large Language Model","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T12:26:37.478169Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2505.24476"},"observation_digest":"sha256:733b6c988d20649329ba6265e5873aafde21adb3d3c21a0b5b80b817576ef269","observation_id":"f13594f2-885c-4c3f-9901-677686589a63","resolution":{"observed_at":"2026-08-07T12:26:37.478169Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T11:59:21.080473Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.00958","last_updated":"2025-06-01T11:07:25Z","snapshot_observed_at":"2026-08-07T11:51:44.408257Z","submitted_at":"2025-06-01T11:07:25Z","title":"Speaking Beyond Language: A Large-Scale Multimodal Dataset for Learning Nonverbal Cues from Video-Grounded Dialogues","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-07T11:59:21.080473Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2506.00958"},"observation_digest":"sha256:dcdcab58ee471a815f8a9c361127636d8dcbc05d5804613f60c3a41a40b8b9b7","observation_id":"db66c58f-dd38-4da6-85ac-4bbd1d2bfdf4","resolution":{"observed_at":"2026-08-07T11:59:21.080473Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T11:52:02.781775Z","title":"Videochat: Chat-centric video understanding.arXiv preprint arXiv:2305.06355, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01300","last_updated":"2025-06-02T04:23:21Z","snapshot_observed_at":"2026-08-10T13:34:27.916880Z","submitted_at":"2025-06-02T04:23:21Z","title":"ReAgent-V: A Reward-Driven Multi-Agent Framework for Video Understanding","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:02.781775Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2506.01300"},"observation_digest":"sha256:f3447bbbeff68d4607831186903f23161ecee6b292d3eea9acacc314381af129","observation_id":"096c6815-5245-45ce-baca-b8e6e4d6cf63","resolution":{"observed_at":"2026-08-07T11:52:02.781775Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T11:51:30.708631Z","title":"Videochat: Chat-centric video understanding,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01307","last_updated":"2025-06-02T04:33:56Z","snapshot_observed_at":"2026-08-10T23:15:32.942485Z","submitted_at":"2025-06-02T04:33:56Z","title":"Align is not Enough: Multimodal Universal Jailbreak Attack against Multimodal Large Language Models","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:30.708631Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2506.01307"},"observation_digest":"sha256:1de46ef5e090994d6b83fb0ee3def9d18de75ca9f2ebe69a8ae5dd74838d66ad","observation_id":"fc6bae13-b3ab-420e-981c-09a913f748a7","resolution":{"observed_at":"2026-08-07T11:51:30.708631Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T11:49:35.768607Z","title":"VideoChat: Chat-centric video understanding,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01364","last_updated":"2025-06-02T06:46:42Z","snapshot_observed_at":"2026-08-08T09:15:38.444850Z","submitted_at":"2025-06-02T06:46:42Z","title":"Unraveling Spatio-Temporal Foundation Models via the Pipeline Lens: A Comprehensive Review","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T11:49:35.768607Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2506.01364"},"observation_digest":"sha256:25753b767ae76c5b901ea8083d460a0428ba2d7e3fcb95d864de0a74adb3bde2","observation_id":"eaedcbe4-f741-478f-a9e9-13c0ef7ab61c","resolution":{"observed_at":"2026-08-07T11:49:35.768607Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T11:46:58.704703Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01520","last_updated":"2025-06-02T10:34:57Z","snapshot_observed_at":"2026-08-10T00:32:07.702364Z","submitted_at":"2025-06-02T10:34:57Z","title":"FormFactory: An Interactive Benchmarking Suite for Multimodal Form-Filling Agents","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T11:46:58.704703Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2506.01520"},"observation_digest":"sha256:3e3af0d691aea840cc8b80e17ea85a2bfb02e0820afcd208b43b2a851a29dd0f","observation_id":"360fda69-e337-43f1-8982-249ddb5c9bf0","resolution":{"observed_at":"2026-08-07T11:46:58.704703Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T11:35:45.945951Z","title":"Videochat: Chat-centric video understanding.arXiv preprint arXiv:2305.06355, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01908","last_updated":"2025-06-02T17:28:26Z","snapshot_observed_at":"2026-08-07T11:29:22.385642Z","submitted_at":"2025-06-02T17:28:26Z","title":"Reinforcement Learning Tuning for VideoLLMs: Reward Design and Data Efficiency","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T11:35:45.945951Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2506.01908"},"observation_digest":"sha256:42d4e8f4ac176aa070c88a804ec363defb185b5847e517e3fbc1f8b51fa712c3","observation_id":"f4f70289-08d2-46c8-ae70-8bcb2c2b402d","resolution":{"observed_at":"2026-08-07T11:35:45.945951Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T12:50:01.880477Z","title":"Videochat: Chat-centric video understanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.03179","last_updated":"2025-05-29T13:17:25Z","snapshot_observed_at":"2026-08-10T21:30:34.921880Z","submitted_at":"2025-05-29T13:17:25Z","title":"Vid-SME: Membership Inference Attacks against Large Video Understanding Models","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T12:50:01.880477Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2506.03179"},"observation_digest":"sha256:f9cc6b3b3d14980619b7f4de81c721d9ddaf642153f62b465acfcf56b6febd28","observation_id":"5affa91c-1097-4287-9971-fc2a790710a4","resolution":{"observed_at":"2026-08-07T12:50:01.880477Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T10:56:05.327297Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.03990","last_updated":"2025-06-04T14:17:42Z","snapshot_observed_at":"2026-08-09T05:23:11.832180Z","submitted_at":"2025-06-04T14:17:42Z","title":"DynTok: Dynamic Compression of Visual Tokens for Efficient and Effective Video Understanding","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-07T10:56:05.327297Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2506.03990"},"observation_digest":"sha256:55cacea76856210456ada2603db268d1f12c4cdc3b86bc3cd2dc224da5a0f873","observation_id":"fb72792c-376e-40b7-80b9-4ed91c6044b2","resolution":{"observed_at":"2026-08-07T10:56:05.327297Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T10:28:52.131174Z","title":"Videochat: Chat-centric video understanding.arXiv preprint arXiv:2305.06355, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05260","last_updated":"2025-06-05T17:21:16Z","snapshot_observed_at":"2026-08-09T11:16:28.299508Z","submitted_at":"2025-06-05T17:21:16Z","title":"LeanPO: Lean Preference Optimization for Likelihood Alignment in Video-LLMs","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:52.131174Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2506.05260"},"observation_digest":"sha256:313078b4361874e9b452c7f250b9a27d575758e84382350731b478172a5e86ed","observation_id":"b5bc3b82-eb86-4917-8094-c69ed13df073","resolution":{"observed_at":"2026-08-07T10:28:52.131174Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T10:29:09.058999Z","title":"Videochat: Chat-centric video understanding.arXiv preprint arXiv:2305.06355, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05287","last_updated":"2025-06-05T17:44:12Z","snapshot_observed_at":"2026-08-09T15:14:29.691315Z","submitted_at":"2025-06-05T17:44:12Z","title":"EOC-Bench: Can MLLMs Identify, Recall, and Forecast Objects in an Egocentric World?","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T10:29:09.058999Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2506.05287"},"observation_digest":"sha256:f4088391787c82002d8d96fd76ada86cb5a9034a0cc9d1ba40132a107d60fb32","observation_id":"2fdc5dd3-0d57-469d-bf4a-92a2918ac617","resolution":{"observed_at":"2026-08-07T10:29:09.058999Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T10:28:50.440440Z","title":"Videochat: Chat-centric video understanding.arXiv preprint arXiv:2305.06355, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05331","last_updated":"2025-06-05T17:59:02Z","snapshot_observed_at":"2026-08-09T02:08:19.234840Z","submitted_at":"2025-06-05T17:59:02Z","title":"MINT-CoT: Enabling Interleaved Visual Tokens in Mathematical Chain-of-Thought Reasoning","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:50.440440Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2506.05331"},"observation_digest":"sha256:d3932474c77312b45df608f635133eabccd6e443351b674d8fc519a93b45fd97","observation_id":"c9dc851b-57e1-45f1-978b-2206b840f82e","resolution":{"observed_at":"2026-08-07T10:28:50.440440Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T10:27:27.739443Z","title":"Videochat: Chat-centric video understanding.arXiv:2305.06355, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05336","last_updated":"2025-07-05T11:38:26Z","snapshot_observed_at":"2026-08-10T11:30:31.815084Z","submitted_at":"2025-06-05T17:59:29Z","title":"VideoMolmo: Spatio-Temporal Grounding Meets Pointing","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T10:27:27.739443Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2506.05336"},"observation_digest":"sha256:6a394b9cc13f2fdbad3c0f26b319f69b1dd2e9157ea50f996362c12b530ba074","observation_id":"f7621811-7c51-449c-ae3c-e6cb18deddc6","resolution":{"observed_at":"2026-08-07T10:27:27.739443Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T10:50:52.645520Z","title":"Videochat: Chat-centric video understanding.arXiv preprint arXiv:2305.06355, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05414","last_updated":"2025-06-04T19:11:20Z","snapshot_observed_at":"2026-08-09T06:09:58.091457Z","submitted_at":"2025-06-04T19:11:20Z","title":"SAVVY: Spatial Awareness via Audio-Visual LLMs through Seeing and Hearing","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T10:50:52.645520Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2506.05414"},"observation_digest":"sha256:71f98ac10c34c7cc56d7a006ae8b714c87d9b9d4cda6c8958e941da9d40bed0b","observation_id":"d7856dab-999d-4b83-a02d-48a4ba13866f","resolution":{"observed_at":"2026-08-07T10:50:52.645520Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T10:20:38.170335Z","title":"Videochat: Chat-centric video understanding.arXiv preprint arXiv:2305.06355, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05689","last_updated":"2025-06-06T02:35:26Z","snapshot_observed_at":"2026-08-09T04:48:26.376897Z","submitted_at":"2025-06-06T02:35:26Z","title":"Pts3D-LLM: Studying the Impact of Token Structure for 3D Scene Understanding With Large Language Models","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:38.170335Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2506.05689"},"observation_digest":"sha256:8de39ef555391bb0b736a470348c20745e75b55d1bc7d3d5756b059df5245e27","observation_id":"1796d316-0616-4672-8205-6ed41d4acc80","resolution":{"observed_at":"2026-08-07T10:20:38.170335Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T10:17:44.040534Z","title":"In Proceedings of the 2023 Conference on Empiri- cal Methods in Natural Language Processing, pages 10462–10479","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05904","last_updated":"2025-06-06T09:23:29Z","snapshot_observed_at":"2026-08-08T22:26:24.817470Z","submitted_at":"2025-06-06T09:23:29Z","title":"Proactive Assistant Dialogue Generation from Streaming Egocentric Videos","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T10:17:44.040534Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2506.05904"},"observation_digest":"sha256:d83011170f33156b2fd56c2cfd2e4981deb3a372993350db6f8ac593484516ec","observation_id":"3f837d44-eabf-4ebd-b8c0-9ed41b93dcd8","resolution":{"observed_at":"2026-08-07T10:17:44.040534Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T06:00:56.898389Z","title":"Videochat: Chat-centric video understanding.arXiv preprint arXiv:2305.06355, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06275","last_updated":"2025-06-06T17:58:36Z","snapshot_observed_at":"2026-08-10T20:35:02.128344Z","submitted_at":"2025-06-06T17:58:36Z","title":"Movie Facts and Fibs (MF$^2$): A Benchmark for Long Movie Understanding","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T06:00:56.898389Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2506.06275"},"observation_digest":"sha256:55eeeb1adc2a0d36be5efe6c4a5b8062a3707e3431060ed286b1e51b2b71a680","observation_id":"d0161f92-a0c5-4a8c-a35f-243b89c77a80","resolution":{"observed_at":"2026-08-07T06:00:56.898389Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T05:49:53.397026Z","title":"Videochat: Chat-centric video understanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.07016","last_updated":"2025-06-13T19:05:47Z","snapshot_observed_at":"2026-08-08T12:14:39.714543Z","submitted_at":"2025-06-08T06:34:29Z","title":"MAGNET: A Multi-agent Framework for Finding Audio-Visual Needles by Reasoning over Multi-Video Haystacks","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T05:49:53.397026Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2506.07016"},"observation_digest":"sha256:016d75ba66467cfa1ef2768d2920e8c51a501b97827eb094a19b713d768f4dda","observation_id":"20fd0780-27b5-4a90-872a-f5ccf4089a5d","resolution":{"observed_at":"2026-08-07T05:49:53.397026Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T05:35:50.832358Z","title":"Videochat: Chat-centric video understanding.arXiv preprint arXiv:2305.06355, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.07600","last_updated":"2025-06-09T10:00:54Z","snapshot_observed_at":"2026-08-09T23:22:18.612266Z","submitted_at":"2025-06-09T10:00:54Z","title":"SceneRAG: Scene-level Retrieval-Augmented Generation for Video Understanding","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T05:35:50.832358Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2506.07600"},"observation_digest":"sha256:4cfe1c776aada68a74b65c95caa1fe3044c6a5142cbb2d9d7db2a693e861133d","observation_id":"c19432be-2e3c-44ab-82f5-9ad99d1453ae","resolution":{"observed_at":"2026-08-07T05:35:50.832358Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T05:31:42.478451Z","title":"Videochat: Chat-centric video understanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.07886","last_updated":"2025-07-19T15:33:13Z","snapshot_observed_at":"2026-08-10T08:12:33.197302Z","submitted_at":"2025-06-09T15:59:25Z","title":"EgoM2P: Egocentric Multimodal Multitask Pretraining","version":3},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T05:31:42.478451Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2506.07886"},"observation_digest":"sha256:7b073233e19504df9de328195d3a5d16a37a82cf4a623d82c3399c5da5df7550","observation_id":"f4bd1b8c-c516-46d8-a784-fb266f7e1a4f","resolution":{"observed_at":"2026-08-07T05:31:42.478451Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T05:27:04.979169Z","title":"Videochat: Chat-centric video understanding.arXiv preprint arXiv:2305.06355, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.07905","last_updated":"2025-06-09T16:20:54Z","snapshot_observed_at":"2026-08-07T21:24:53.935343Z","submitted_at":"2025-06-09T16:20:54Z","title":"WeThink: Toward General-purpose Vision-Language Reasoning via Reinforcement Learning","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T05:27:04.979169Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2506.07905"},"observation_digest":"sha256:aad19488a8e8b4debbcd57899c8509befc648f010729438161da69cb5ba4e8a2","observation_id":"43752052-9380-4f73-bef0-41ec3dc11c36","resolution":{"observed_at":"2026-08-07T05:27:04.979169Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T04:54:22.296324Z","title":"Videochat: Chat-centric video understanding.arXiv preprint arXiv:2305.06355, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.09445","last_updated":"2025-06-11T06:52:31Z","snapshot_observed_at":"2026-08-11T04:53:31.748297Z","submitted_at":"2025-06-11T06:52:31Z","title":"TOGA: Temporally Grounded Open-Ended Video QA with Weak Supervision","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T04:54:22.296324Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2506.09445"},"observation_digest":"sha256:2437e61a2ba22967307b6092ba5431cb99ca54fec81181040203faa48dc3c12a","observation_id":"dafab7fb-eddf-4b09-a8f5-610c60bdfd5c","resolution":{"observed_at":"2026-08-07T04:54:22.296324Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T04:40:43.345468Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.09953","last_updated":"2025-06-11T17:23:35Z","snapshot_observed_at":"2026-08-07T09:14:08.520195Z","submitted_at":"2025-06-11T17:23:35Z","title":"Outside Knowledge Conversational Video (OKCV) Dataset -- Dialoguing over Videos","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-07T04:40:43.345468Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2506.09953"},"observation_digest":"sha256:573b703b7f4ba7c0596359be24a8cd1254fa6daac532de6296e1c03589ca3294","observation_id":"0bca8184-4516-4ef8-9cd2-32287af706f8","resolution":{"observed_at":"2026-08-07T04:40:43.345468Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T00:41:51.274375Z","title":"Videochat: Chat-centric video understanding.arXiv preprint arXiv:2305.06355, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.12992","last_updated":"2025-06-15T23:20:08Z","snapshot_observed_at":"2026-08-10T13:05:33.003989Z","submitted_at":"2025-06-15T23:20:08Z","title":"SmartHome-Bench: A Comprehensive Benchmark for Video Anomaly Detection in Smart Homes Using Multi-Modal Large Language Models","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T00:41:51.274375Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2506.12992"},"observation_digest":"sha256:43c79ef1f35693ffff07c393ce1db0a83d9a3d31cab38743a9a1e10564ec6972","observation_id":"382a5726-de98-4256-b7eb-0cd9cc1efef9","resolution":{"observed_at":"2026-08-07T00:41:51.274375Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T00:33:19.316833Z","title":"Videochat: Chat-centric video understanding, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.13642","last_updated":"2025-06-22T07:56:58Z","snapshot_observed_at":"2026-08-10T10:57:00.100394Z","submitted_at":"2025-06-16T16:06:45Z","title":"Stream-Omni: Simultaneous Multimodal Interactions with Large Language-Vision-Speech Model","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T00:33:19.316833Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2506.13642"},"observation_digest":"sha256:a36bb692b0941a60955ce3cb1dc4379b5519883750ec994a7cae4c8ad983e308","observation_id":"bd4ad7ce-5871-44c9-9b2f-0bc008ee470e","resolution":{"observed_at":"2026-08-07T00:33:19.316833Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-06T23:49:30.713603Z","title":"Videochat: Chat-centric video understanding,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.16082","last_updated":"2025-06-19T07:07:42Z","snapshot_observed_at":"2026-08-09T11:41:47.683908Z","submitted_at":"2025-06-19T07:07:42Z","title":"PR-DETR: Injecting Position and Relation Prior for Dense Video Captioning","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-06T23:49:30.713603Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2506.16082"},"observation_digest":"sha256:081c609e1e74803be00975ad810e03283c1a4c5d8b3c9c4cee83b69d65b35697","observation_id":"d0abc37a-0e0c-421d-b03e-a92d76fd7b81","resolution":{"observed_at":"2026-08-06T23:49:30.713603Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-06T23:34:52.248369Z","title":"Videochat: Chat-centric video understanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.17545","last_updated":"2025-06-21T02:29:10Z","snapshot_observed_at":"2026-08-10T21:22:23.494902Z","submitted_at":"2025-06-21T02:29:10Z","title":"Scene-R1: Video-Grounded Large Language Models for 3D Scene Reasoning without 3D Annotations","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T23:34:52.248369Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2506.17545"},"observation_digest":"sha256:32f5ded433d933bf3f14be9a2890ca5fd4ed639302b4031fb67f846f87f6d0f3","observation_id":"5ff9199b-effe-478f-8564-e0160dc4a7d3","resolution":{"observed_at":"2026-08-06T23:34:52.248369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-06T23:29:29.218410Z","title":"VideoChat: Chat-centric video understanding,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.18071","last_updated":"2025-06-27T06:32:43Z","snapshot_observed_at":"2026-08-08T12:15:56.187844Z","submitted_at":"2025-06-22T15:39:02Z","title":"MUPA: Towards Multi-Path Agentic Reasoning for Grounded Video Question Answering","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T23:29:29.218410Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2506.18071"},"observation_digest":"sha256:882954d1ff3270009d595a85a98e2f077255834b5a1fe55c4327726a67278e4b","observation_id":"cc5d28cb-ac01-4903-b3b0-4cb656dca795","resolution":{"observed_at":"2026-08-06T23:29:29.218410Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2305.06355/citation-record","integrity":"/paper/2305.06355/integrity","json":"/paper/2305.06355/citation-record.json","paper":"/paper/2305.06355"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Openflamingo, March 2023","venue":null,"work_id":"fe04ce32-7667-49a6-912b-ca2a272d8271","year":2023},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:d7ad8c1fb1e77029e756c6b07a637fbfc8c41d09875a5a2242e7b34703ce71ed","observation_id":"4f1d7275-bcfb-42f1-8244-508f8918f414","resolution":{"observed_at":"2026-05-13T23:30:00.809695Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Frozen in time: A joint video and image encoder for end-to-end retrieval","venue":null,"work_id":"14d60c0b-f95c-4e49-877b-d7529b229ac7","year":2021},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:be4c606c90a70801b2b0004077e90b1e0d716bb525f21505fd24354948e86400","observation_id":"243c31df-d978-462a-b328-eee78fd535be","resolution":{"observed_at":"2026-05-13T23:30:00.749928Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Language models are few-shot learners","venue":null,"work_id":"06921215-168b-4266-a8bd-53d84ad473f0","year":1901},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:c92900d53e41a713f28c3551900293178e390d0019bbe511a5f36f8c6af3f944","observation_id":"0b04bd30-56ac-4b1e-97a0-6c38aacacf6b","resolution":{"observed_at":"2026-05-13T23:30:00.754276Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Conceptual 12m: Pushing web-scale image-text pre-training to recognize long-tail visual concepts","venue":null,"work_id":"5d3c3db1-3a21-4551-8576-912217ab41b4","year":2021},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:4a5b0bc4a5e68f32096bd0977978f09285d404cc166f8d34d86422599d0403e0","observation_id":"38cbb19d-6359-4c6a-8cbd-94ac5dabec98","resolution":{"observed_at":"2026-05-13T23:30:00.758413Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2211.09529","last_updated":"2022-11-17T13:45:06Z","snapshot_observed_at":"2026-08-10T05:30:20.599642Z","submitted_at":"2022-11-17T13:45:06Z","title":"InternVideo-Ego4D: A Pack of Champion Solutions to Ego4D Challenges","version":1},"cited_work":{"arxiv_id":"2211.09529","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2211.09529","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Internvideo-ego4d: A pack of champion solutions to ego4d challenges","venue":null,"work_id":"54773a77-a0f2-43d2-9158-858d1d0652f9","year":2022},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/2211.09529","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:42105826f144842850d371b09adacd0c220ff4d629ec30c3d1ade68ac0066cdc","observation_id":"d6438f78-fafb-4231-869f-24ece27e1510","resolution":{"observed_at":"2026-05-13T23:30:00.504755Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1504.00325","last_updated":"2015-04-03T20:21:16Z","snapshot_observed_at":"2026-08-04T18:05:27.145522Z","submitted_at":"2015-04-01T18:13:43Z","title":"Microsoft COCO Captions: Data Collection and Evaluation Server","version":2},"cited_work":{"arxiv_id":"1504.00325","doi":"10.48550/arxiv.1504.00325","metadata_source":"pith","pith_arxiv_id":"1504.00325","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microsoft COCO Captions: Data Collection and Evaluation Server","venue":"cs.CV","work_id":"b3d6fb46-4169-4a28-8f7e-2ca6774211da","year":2015},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/1504.00325","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:c1a624448e85aaa022f845947b140389f499204ad1bdccb01f943cae84fe6c3d","observation_id":"e3d6e35d-5704-4e6a-80c4-1fd9563f1deb","resolution":{"observed_at":"2026-05-13T23:30:00.510458Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vicuna: An open-source chatbot impressing gpt-4 with 90%* chatgpt quality","venue":null,"work_id":"c1a3009a-3b02-4598-9772-07e8059a3470","year":2023},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:0d42505ac79bc400fbd8e19f89098d734139f7dbdf7feda5a8c5f0e3ed144171","observation_id":"f8fcdbca-4181-4598-b423-7a3f47e98c71","resolution":{"observed_at":"2026-05-13T23:30:00.769975Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.11416","last_updated":"2022-12-06T21:39:48Z","snapshot_observed_at":"2026-07-06T14:08:18.855958Z","submitted_at":"2022-10-20T16:58:32Z","title":"Scaling Instruction-Finetuned Language Models","version":5},"cited_work":{"arxiv_id":"2210.11416","doi":"10.48550/arxiv.2210.11416","metadata_source":"pith","pith_arxiv_id":"2210.11416","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Scaling Instruction-Finetuned Language Models","venue":"cs.LG","work_id":"8405abb1-7558-4fdf-af24-f4c52fa77a06","year":2022},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/2210.11416","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:e39d83d33e19a7911c222c1b02c647c00d32c364a66cacda58427c5a5a368a52","observation_id":"5a610f2f-ed84-41c7-ad5b-5ff2f14d57fa","resolution":{"observed_at":"2026-05-13T23:30:00.519803Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"764e76f4-81f0-4fed-8779-93c10ea1a0a5","year":2023},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:a1fb6579854cded05af8c2f98568298d6c46598570e36a4a1ad0bc6d8d8fa38d","observation_id":"06c4edb0-be53-4a0a-a1e8-1d2d72fec63e","resolution":{"observed_at":"2026-05-13T23:30:00.778125Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Stablelm: Stability ai language models","venue":null,"work_id":"b20c6087-e981-4e20-aefb-e510ef6f402f","year":2023},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:8277cb9f5468561dc40364e3bf4d5137b268f6f341982ffc1c582616414687c3","observation_id":"448d3caa-667f-4136-bec8-b84ba513e11e","resolution":{"observed_at":"2026-05-13T23:30:00.781707Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"An empirical study of training end-to-end vision-and- language transformers","venue":null,"work_id":"e298ce0c-2367-4a7b-a894-9dd3c754981a","year":2022},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:43ba274e3105e45715fb57227fdf36609f2d20434b04993dc3dedcd9b8b694b5","observation_id":"df5dde4e-4fda-4135-b59e-247aacefc6ec","resolution":{"observed_at":"2026-05-13T23:30:00.785457Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.12681","last_updated":"2022-04-16T04:21:26Z","snapshot_observed_at":"2026-08-04T04:22:15.975890Z","submitted_at":"2021-11-24T18:31:20Z","title":"VIOLET : End-to-End Video-Language Transformers with Masked Visual-token Modeling","version":2},"cited_work":{"arxiv_id":"2111.12681","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2111.12681","snapshot_observed_at":"2026-07-03T00:57:30.204945Z","title":"Violet: End-to-end video-language transformers with masked visual-token modeling","venue":null,"work_id":"c7602525-8155-4a61-bd4e-abf06c26e6e7","year":2021},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/2111.12681","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:cf9744e55eccc3fe38985fa0cbf37b1356fcba4855801c33b3cc1c22ce5c3848","observation_id":"ab787156-95d0-45d3-be18-1de421898834","resolution":{"observed_at":"2026-05-13T23:30:00.538637Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Scaling up vision-language pre-training for image captioning","venue":null,"work_id":"fe9157dd-28bf-441a-b40f-5f07530cd3c1","year":2022},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:45993eb436f8bae74622aafa2c57007aa9e86bf024090b950c3c74aa1686ca74","observation_id":"16a193ea-9225-410a-a0c7-982b383dc647","resolution":{"observed_at":"2026-05-13T23:30:00.794460Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.14045","last_updated":"2023-03-01T11:04:51Z","snapshot_observed_at":"2026-08-08T05:52:17.343895Z","submitted_at":"2023-02-27T18:55:27Z","title":"Language Is Not All You Need: Aligning Perception with Language Models","version":2},"cited_work":{"arxiv_id":"2302.14045","doi":null,"metadata_source":"pith","pith_arxiv_id":"2302.14045","snapshot_observed_at":"2026-07-04T19:20:06.296460Z","title":"Language Is Not All You Need: Aligning Perception with Language Models","venue":"cs.CL","work_id":"2a1e0563-79f5-4521-8293-b8b1aebf7cee","year":2023},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/2302.14045","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:4be5d36abf5ca9cb5fda4176d46e70b86274c5972a546631f28488bf29c3275c","observation_id":"4a99b478-b6f0-497d-913d-cb6931eee182","resolution":{"observed_at":"2026-05-15T18:32:23.026112Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.05657","last_updated":"2024-03-18T12:34:30Z","snapshot_observed_at":"2026-07-06T15:01:06.236331Z","submitted_at":"2023-03-10T02:16:35Z","title":"Tag2Text: Guiding Vision-Language Model via Image Tagging","version":3},"cited_work":{"arxiv_id":"2303.05657","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2303.05657","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Tag2text: Guiding vision-language model via image tagging","venue":null,"work_id":"9f16058a-f12c-4bbc-878c-6650d9acc38c","year":2023},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/2303.05657","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:3f1511a94c1d125b1ca2ed6b101779f964a8f6eb1ae2164004585e2589a10da8","observation_id":"5d6d4010-0a8e-4b89-a0fd-e76c14b4cc0b","resolution":{"observed_at":"2026-05-13T23:30:00.498208Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Dolphin: General video interaction platform based on llms","venue":null,"work_id":"058441ba-0202-4f7d-a336-cb8ce2fa7571","year":2023},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:9a95e6397385fb9cc622ad1b4bca5c9ea8c22778ced8ee295eb5af587244eabb","observation_id":"90bcb349-2d2f-40f7-8277-c3e4861f24d1","resolution":{"observed_at":"2026-05-13T23:30:00.814032Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Visual genome: Connecting language and vision using crowdsourced dense image annotations","venue":null,"work_id":"c65502dd-192e-483a-9ab3-350bdfc9c6d1","year":2017},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:511d1e6804c4b56cd43178cc2966c8092005d3cae8d83a34202a6e024f34f0db","observation_id":"6fc4f452-f14b-4c0a-8de7-91bd62e3ea64","resolution":{"observed_at":"2026-05-13T23:30:00.821325Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.12597","last_updated":"2023-06-15T07:57:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-01-30T00:56:51Z","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","version":3},"cited_work":{"arxiv_id":"2301.12597","doi":"10.48550/arxiv.2301.12597","metadata_source":"pith","pith_arxiv_id":"2301.12597","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","venue":"cs.CV","work_id":"63d03f4d-15f4-4583-8286-913c19f02294","year":2023},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/2301.12597","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:08728551e0e8cec10bb24937316147827063ba03e6f3e7774e6542d9e31a7d7f","observation_id":"072ee880-f327-46cd-82a0-9d9f90a051b9","resolution":{"observed_at":"2026-05-13T23:30:00.555232Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-07-12T21:49:47.354893+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T21:49:47.354893+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation","venue":null,"work_id":"c409f9a1-c03d-4398-a7a4-b847d5e3c145","year":2022},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:3cfab135c27fde6fb680f5812c9356a06d8f5362909d7e22d01271611784773d","observation_id":"dc230461-5565-47ef-92c1-357270b0a820","resolution":{"observed_at":"2026-05-13T23:30:00.830565Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2211.09552","last_updated":"2022-11-17T14:17:40Z","snapshot_observed_at":"2026-08-07T04:39:25.258902Z","submitted_at":"2022-11-17T14:17:40Z","title":"UniFormerV2: Spatiotemporal Learning by Arming Image ViTs with Video UniFormer","version":1},"cited_work":{"arxiv_id":"2211.09552","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2211.09552","snapshot_observed_at":"2026-07-03T19:08:49.809225Z","title":"Uniformerv2: Spatiotemporal learning by arming image vits with video uniformer","venue":null,"work_id":"cb4c2612-0c2a-4ab9-9dde-7031260274d1","year":2022},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/2211.09552","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:4e3c97ce74a5e6db563533f786e1d7301b4fdd86644a37d14b932aa0b976ef2c","observation_id":"a5b874da-8fef-43d5-8329-ae9d9f30a5a9","resolution":{"observed_at":"2026-05-13T23:30:00.561864Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.16058","last_updated":"2024-03-11T09:21:50Z","snapshot_observed_at":"2026-07-06T15:09:01.194260Z","submitted_at":"2023-03-28T15:39:28Z","title":"Unmasked Teacher: Towards Training-Efficient Video Foundation Models","version":2},"cited_work":{"arxiv_id":"2303.16058","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2303.16058","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"arXiv preprint arXiv:2303.16058 , year=","venue":null,"work_id":"28bed721-6c21-4176-b6fe-0a81b63f75e3","year":2023},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/2303.16058","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:f84e9077ea38e6acfb4e38e20374fdd35e419285b5d8158deb52b8a1c5305740","observation_id":"ab5c4fc4-b242-4bda-93f8-b7df99450714","resolution":{"observed_at":"2026-05-13T23:30:00.572703Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.07160","last_updated":"2022-06-14T20:43:25Z","snapshot_observed_at":"2026-07-06T13:20:51.362304Z","submitted_at":"2022-06-14T20:43:25Z","title":"LAVENDER: Unifying Video-Language Understanding as Masked Language Modeling","version":1},"cited_work":{"arxiv_id":"2206.07160","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2206.07160","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Lavender: Unifying video-language understanding as masked language modeling","venue":null,"work_id":"a66881b8-36b2-40f7-94ff-15730443b004","year":2022},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/2206.07160","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:b167fe0b3e007fe478f2e1b415d32bd0b0a1eb99996de9ccdfbc3d221d33a44e","observation_id":"95867805-c515-4620-b73c-0fb3e2bf0c0a","resolution":{"observed_at":"2026-05-13T23:30:00.578691Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2001.05691","last_updated":"2021-01-28T01:43:34Z","snapshot_observed_at":"2026-08-10T17:29:43.107061Z","submitted_at":"2020-01-16T08:28:57Z","title":"Learning Spatiotemporal Features via Video and Text Pair Discrimination","version":3},"cited_work":{"arxiv_id":"2001.05691","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2001.05691","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Learning spatiotemporal fea- tures via video and text pair discrimination","venue":null,"work_id":"f9fa1d6c-64f0-4b92-bdb7-a4a98296fd68","year":2001},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/2001.05691","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:06071fd0e49fb5dedbf25a0d7cd8c72def98cda26a780e9d31e4b01faf635544","observation_id":"78b1790f-89ba-42ad-9701-6e8853b2cb3b","resolution":{"observed_at":"2026-05-13T23:30:00.585193Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.16434","last_updated":"2023-03-29T03:30:38Z","snapshot_observed_at":"2026-07-06T15:09:21.780147Z","submitted_at":"2023-03-29T03:30:38Z","title":"TaskMatrix.AI: Completing Tasks by Connecting Foundation Models with Millions of APIs","version":1},"cited_work":{"arxiv_id":"2303.16434","doi":"10.48550/arxiv.2303.16434","metadata_source":"arxiv_reference","pith_arxiv_id":"2303.16434","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Taskmatrix.ai: Completing tasks by connecting foundation models with millions of apis","venue":"arXiv (Cornell University)","work_id":"90f98bad-b70c-481c-a56f-1197ee89e441","year":2023},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/2303.16434","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:55fec9d12a723522c8f5df0effa3ab20088fd710fdcce4570f258b838d75369e","observation_id":"93260ef7-5c24-485d-93a9-050fee0c7104","resolution":{"observed_at":"2026-05-13T23:30:00.591327Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Visual instruction tuning","venue":null,"work_id":"b4e7d21d-d1c1-4cef-acf5-683f1bd9cde4","year":2023},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:02d0bd3d38f9cba0d87f92cf538ad96cc90dd6f05a34761c36cde1b786f22b20","observation_id":"d5d1a375-3258-4c0a-8cce-5d2ae4dcc3a5","resolution":{"observed_at":"2026-05-13T23:30:00.857147Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":"2305.05662","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-07-03T20:08:55.658612Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond lan- guage","venue":null,"work_id":"456c399b-9545-4252-b591-33d7b091fbf0","year":2023},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:fa897ae129e5c67948871805c6b00c0c74cdd8f49c9467ffd572edce58838af1","observation_id":"dd337dbf-d56d-4135-a28d-da0da5248840","resolution":{"observed_at":"2026-05-13T23:30:00.598826Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"End-to-end learning of visual representations from uncurated instructional videos","venue":null,"work_id":"baa1286d-3fc1-41f9-a3f8-c9a951999879","year":2020},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:f76839b9ac3807e123db2dce2d1a35510d156bd926058e91bd28e705bc5a8ec2","observation_id":"8fa12758-6454-4ca9-93b2-12ee0c0b55f7","resolution":{"observed_at":"2026-05-13T23:30:00.866241Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.08773","last_updated":"2022-03-14T09:15:08Z","snapshot_observed_at":"2026-08-05T18:13:22.401467Z","submitted_at":"2021-04-18T08:44:56Z","title":"Cross-Task Generalization via Natural Language Crowdsourcing Instructions","version":4},"cited_work":{"arxiv_id":"2104.08773","doi":"10.1002/wps.20513","metadata_source":"pith","pith_arxiv_id":"2104.08773","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Cross-Task Generalization via Natural Language Crowdsourcing Instructions","venue":"cs.CL","work_id":"807e5cef-0bb3-4eae-a684-154fc2d6aa06","year":2021},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/2104.08773","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:2a102071f1395be94dc05cb17c31f0fe52c332fbca1bffed97fab7f20687a08d","observation_id":"b3b60760-69fe-4464-a775-e10d6d0dfea3","resolution":{"observed_at":"2026-05-18T01:57:29.846007Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gpt-4 technical report","venue":null,"work_id":"9c3abadc-61e1-4f07-b2b4-38c913ffb68f","year":2023},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:d35b4cbdaf15ea82b0dab4b758fca669cae1730f6f5ada2537883364656d28bd","observation_id":"fb55e015-b5ab-44a4-8637-d8ecb7375671","resolution":{"observed_at":"2026-05-13T23:30:00.873300Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Chatgpt: Optimizing language models for dialogue","venue":null,"work_id":"cde5c8d2-bb59-4bba-a459-1635b63e7d3f","year":2022},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:8f007c326f4dddec9202fbeff3316f738eabb478e921e3b3f62f956a76fb1dda","observation_id":"3fdb47b0-ef44-4665-a138-0980a99c4e32","resolution":{"observed_at":"2026-05-13T23:30:00.878360Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Im2text: Describing images using 1 million captioned photographs","venue":null,"work_id":"2e6606d0-10bb-4137-82dd-b798d6315c12","year":2011},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:323e1df6b079822f1a8268b0cee97873a5f5b3dda9dca73dc4f1433a52f65822","observation_id":"c8762e9d-e69c-4a4d-a13b-fc8b84b1dbec","resolution":{"observed_at":"2026-05-13T23:30:00.762228Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Training language models to follow instructions with human feedback","venue":null,"work_id":"22a399d7-c4ba-4344-86b9-7885598b17ce","year":2022},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:7b516376389f1a6fbbd08f27a823a678bd56d671a23ddadcfc97ffcd80b7068e","observation_id":"37268e7b-c1b3-4473-ac6f-e599e20b9bed","resolution":{"observed_at":"2026-05-13T23:30:00.766064Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.04356","last_updated":"2022-12-06T18:46:04Z","snapshot_observed_at":"2026-07-06T14:28:21.844826Z","submitted_at":"2022-12-06T18:46:04Z","title":"Robust Speech Recognition via Large-Scale Weak Supervision","version":1},"cited_work":{"arxiv_id":"2212.04356","doi":"10.48550/arxiv.2212.04356","metadata_source":"pith","pith_arxiv_id":"2212.04356","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Robust Speech Recognition via Large-Scale Weak Supervision","venue":"eess.AS","work_id":"ed5fbefe-24bf-436f-98c1-68e9114360bf","year":2022},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/2212.04356","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:7679de49429c55f5b36bc6c3d6fb6059119573b0b7936bb5b18477e7441f1e53","observation_id":"1f5592a9-5b6c-4a81-bcce-e25bde195f02","resolution":{"observed_at":"2026-05-13T23:30:00.609860Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-03T20:38:04.271886+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T20:38:04.271886+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Exploring the limits of transfer learning with a unified text-to-text transformer","venue":null,"work_id":"88eb9e91-ade3-4114-81a5-f669d50946ea","year":2020},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:41029293dcfe43ebccc87f76e0900a7bee17f0a169a9cfe7794e5af4b625acfb","observation_id":"837ee617-26ac-4b68-91ef-73d73f32acd8","resolution":{"observed_at":"2026-05-13T23:30:00.789810Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Conceptual captions: A cleaned, hypernymed, image alt-text dataset for automatic image captioning","venue":null,"work_id":"820d7f2b-ab0a-4a00-938c-46cad7bf8432","year":2018},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:a0c4c06f2cc361ae1143b6962ba2afbc43283043bdf22005449d01e2f2bb71e4","observation_id":"b18076b8-4fd7-4dcd-90da-eb9297708857","resolution":{"observed_at":"2026-05-13T23:30:00.801744Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2107.06383","last_updated":"2021-07-13T20:48:12Z","snapshot_observed_at":"2026-08-07T22:40:13.912253Z","submitted_at":"2021-07-13T20:48:12Z","title":"How Much Can CLIP Benefit Vision-and-Language Tasks?","version":1},"cited_work":{"arxiv_id":"2107.06383","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2107.06383","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"How much can clip beneﬁt vision-and-language tasks?","venue":null,"work_id":"3b4c7819-6f34-4c2d-91df-038dc5d7c576","year":2021},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/2107.06383","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:cb300881f8be70605e37568607ff6704dae8986a13156a257bccaec35c030d49","observation_id":"7825d19f-eb53-4f0d-bfd3-6f58d4a5bcf9","resolution":{"observed_at":"2026-05-13T23:30:00.616454Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.17580","last_updated":"2023-12-03T18:17:21Z","snapshot_observed_at":"2026-08-03T00:52:54.308486Z","submitted_at":"2023-03-30T17:48:28Z","title":"HuggingGPT: Solving AI Tasks with ChatGPT and its Friends in Hugging Face","version":4},"cited_work":{"arxiv_id":"2303.17580","doi":"10.48550/arxiv.2303.17580","metadata_source":"pith","pith_arxiv_id":"2303.17580","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"HuggingGPT: Solving AI Tasks with ChatGPT and its Friends in Hugging Face","venue":"cs.CL","work_id":"f20ed1da-2676-4598-a11b-54549718735b","year":2023},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/2303.17580","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:d01cde2fee6f3fc9c6b1b4e991b1116f6015bd22ac96de786b6e287b210d0a9e","observation_id":"74912950-f8ee-4edc-b5f1-af1b1f977abf","resolution":{"observed_at":"2026-05-14T00:06:45.816188Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-07-12T03:19:34.235168+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T03:19:34.235168+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Murphy, and Cordelia Schmid","venue":null,"work_id":"bd228181-4788-4533-a1ed-ee050e8ab84b","year":2019},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:ff46ef9e802003622dae868b367d19e288ba5ebaff03f73d2d0b4df6abe03277","observation_id":"1a772ef6-6cb4-483e-925b-dac2e4448560","resolution":{"observed_at":"2026-05-13T23:30:00.839773Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.15389","last_updated":"2023-03-27T17:02:21Z","snapshot_observed_at":"2026-07-06T15:08:34.018146Z","submitted_at":"2023-03-27T17:02:21Z","title":"EVA-CLIP: Improved Training Techniques for CLIP at Scale","version":1},"cited_work":{"arxiv_id":"2303.15389","doi":"10.48550/arxiv.2303.15389","metadata_source":"pith","pith_arxiv_id":"2303.15389","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"EVA-CLIP: Improved Training Techniques for CLIP at Scale","venue":"cs.CV","work_id":"0c16c250-fd0f-446a-bbb0-ea8dd0ba5ccd","year":2023},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/2303.15389","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:c89a20c800abe8b1dc41625c10c04a9bd74901161a98f99785e7118d3645561e","observation_id":"e12117b1-6c89-4e1c-a853-cc27b8436508","resolution":{"observed_at":"2026-05-13T23:30:00.627028Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Hashimoto","venue":null,"work_id":"a0d4d607-12ad-4098-b8f8-6db667b8ac09","year":2023},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:479ca9a9f6591740a2b3b1e2ce8a096c73b72138792b4a44fce12ac5eb609110","observation_id":"34774cd3-7749-47dd-b53b-e80404f37492","resolution":{"observed_at":"2026-05-13T23:30:00.850477Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Videomae: Masked autoencoders are data-efficient learners for self-supervised video pre-training","venue":null,"work_id":"1f374364-6039-4ee7-9686-2e02e38a1b5f","year":2022},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:bc105b50de5f4cbac1ce20d4bfde77b3da0dc97685de78bc780ef6af10b5b4b4","observation_id":"8837ab19-0e7f-4e4c-9102-0676edff31b7","resolution":{"observed_at":"2026-05-13T23:30:00.854067Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":"2302.13971","doi":"10.48550/arxiv.2302.13971","metadata_source":"pith","pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LLaMA: Open and Efficient Foundation Language Models","venue":"cs.CL","work_id":"c018fc23-6f3f-4035-9d02-28a2173b2b9d","year":2023},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:05f56f3bd8872a944dce93993f443c329b84886556f2994dce35af90e9eb303c","observation_id":"c1539f79-b1e3-49b0-a27e-acefe3128c87","resolution":{"observed_at":"2026-05-13T23:30:00.639179Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-08T16:08:17.350515+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T16:08:17.350515+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.07303","last_updated":"2022-03-14T17:06:30Z","snapshot_observed_at":"2026-07-06T12:47:38.925428Z","submitted_at":"2022-03-14T17:06:30Z","title":"All in One: Exploring Unified Video-Language Pre-training","version":1},"cited_work":{"arxiv_id":"2203.07303","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2203.07303","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"All in one: Exploring uniﬁed video-language pre-training","venue":null,"work_id":"91bd00d9-4041-47ad-a58d-043b0c543c9c","year":2022},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/2203.07303","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:3ac660c5c4f160ed4806d4c6bc1f70b5b9b00564c2b210926853fb8a845f0027","observation_id":"3a02762d-64c5-4419-932f-1de2d9e32d0a","resolution":{"observed_at":"2026-05-13T23:30:00.646028Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Videomae v2: Scaling video masked autoencoders with dual masking","venue":null,"work_id":"ad2beb12-72aa-4df1-ba20-8393af9d3813","year":2023},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:e7a8653f88b59b32e16b8204fb8fb7a961d8c048781c7fb7c94ec236cd1a93c1","observation_id":"8289f058-a571-428a-8d0a-0d3825f37c8d","resolution":{"observed_at":"2026-05-13T23:30:00.773726Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Internimage: Exploring large-scale vision foundation models with deformable convolutions","venue":null,"work_id":"b7218252-b741-4206-a191-7b16ea3e057a","year":2023},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:41ec9d968b1b4af07bcb6f396aa637612a5554a93b762a92d98804a875e9d8a7","observation_id":"5187eb9a-34b1-494a-9c95-6e71d691d44c","resolution":{"observed_at":"2026-05-13T23:30:00.826143Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.03191","last_updated":"2022-12-07T12:20:55Z","snapshot_observed_at":"2026-07-06T14:27:34.639236Z","submitted_at":"2022-12-06T18:09:49Z","title":"InternVideo: General Video Foundation Models via Generative and Discriminative Learning","version":2},"cited_work":{"arxiv_id":"2212.03191","doi":"10.48550/arxiv.2212.03191","metadata_source":"pith","pith_arxiv_id":"2212.03191","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVideo: General Video Foundation Models via Generative and Discriminative Learning","venue":"cs.CV","work_id":"780aaeee-ac26-46b1-b6ff-64a7a624e694","year":2022},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/2212.03191","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:fea6eeca6d2de47d6ba605c10b48e38a5e6b84c2fde11fdcd395222921aad2ce","observation_id":"01890349-770e-4c2c-9b11-0c15604e7c9d","resolution":{"observed_at":"2026-05-17T00:36:53.574468Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.04671","last_updated":"2023-03-08T15:50:02Z","snapshot_observed_at":"2026-07-06T15:00:13.368355Z","submitted_at":"2023-03-08T15:50:02Z","title":"Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models","version":1},"cited_work":{"arxiv_id":"2303.04671","doi":"10.48550/arxiv.2303.04671","metadata_source":"pith","pith_arxiv_id":"2303.04671","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models","venue":"cs.CV","work_id":"b06ebfb8-5543-4f2c-af49-c05c4e63fc45","year":2023},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/2303.04671","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:a9ba761591f4665306302140d3fbb447b5ed2ef3fc29d5fb238b9bd35392d23e","observation_id":"71cca06b-0916-4d52-a063-8629de8e0e6c","resolution":{"observed_at":"2026-05-13T23:30:00.663677Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.00280","last_updated":"2022-12-01T04:59:44Z","snapshot_observed_at":"2026-07-06T14:25:28.974939Z","submitted_at":"2022-12-01T04:59:44Z","title":"GRiT: A Generative Region-to-text Transformer for Object Understanding","version":1},"cited_work":{"arxiv_id":"2212.00280","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2212.00280","snapshot_observed_at":"2026-07-02T02:26:26.561841Z","title":"Grit: A generative region-to-text transformer for object understanding","venue":null,"work_id":"f943f4a1-5ff5-4ce2-b810-35e7e94beee3","year":2022},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/2212.00280","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:ed333b1f614a6fa46ed52f35fd18584d7eb4d97535fc1ea9bc6d57f8c032f0c6","observation_id":"52ccaa0b-309f-4256-bf89-57f0f0adba60","resolution":{"observed_at":"2026-05-13T23:30:00.669366Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.14084","last_updated":"2021-10-01T15:13:27Z","snapshot_observed_at":"2026-08-04T06:58:39.755482Z","submitted_at":"2021-09-28T23:01:51Z","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","version":2},"cited_work":{"arxiv_id":"2109.14084","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.14084","snapshot_observed_at":"2026-07-04T06:39:37.380967Z","title":"Videoclip: Contrastive pre-training for zero-shot video-text understanding","venue":null,"work_id":"509d3b12-48dd-4984-9d28-a28d917f0b1c","year":2021},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/2109.14084","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:f89ea8d8d9914332ca39398c0c227bde16e67f009cdc1ddc0a43cde2acf8ac90","observation_id":"b9b1d4cd-aa91-48eb-b583-b569e1e5952e","resolution":{"observed_at":"2026-05-13T23:30:00.675840Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.11381","last_updated":"2023-03-20T18:31:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-20T18:31:47Z","title":"MM-REACT: Prompting ChatGPT for Multimodal Reasoning and Action","version":1},"cited_work":{"arxiv_id":"2303.11381","doi":"10.48550/arxiv.2303.11381","metadata_source":"pith","pith_arxiv_id":"2303.11381","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MM-REACT: Prompting ChatGPT for Multimodal Reasoning and Action","venue":"cs.CV","work_id":"6dc43db8-227d-438e-8658-0c8acecba08a","year":2023},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/2303.11381","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:a72636b48fcaa5cfaadf19406c3c517b6484fbe8c4bd342e85930e8e68be1209","observation_id":"013852b1-7a80-4f7e-b826-b5e1872d8a59","resolution":{"observed_at":"2026-05-14T01:17:58.977376Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.07783","last_updated":"2021-11-09T17:15:38Z","snapshot_observed_at":"2026-07-06T12:08:35.761426Z","submitted_at":"2021-11-09T17:15:38Z","title":"FILIP: Fine-grained Interactive Language-Image Pre-Training","version":1},"cited_work":{"arxiv_id":"2111.07783","doi":null,"metadata_source":"pith","pith_arxiv_id":"2111.07783","snapshot_observed_at":"2026-07-11T03:07:53.330969Z","title":"Filip: Fine-grained interactive language-image pre-training","venue":"cs.CV","work_id":"ed22ea81-4b1d-4b6e-96fc-aa013a968e9a","year":2021},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/2111.07783","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:14ee22ea345c76772648e74ac9a715de1aa82ba5004229f81e5ec313d9631a16","observation_id":"25df6251-3b0c-4dbc-8297-b07522936587","resolution":{"observed_at":"2026-05-13T23:30:00.687974Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"mplug-owl: Modularization empowers large language models with multimodality","venue":null,"work_id":"f7285cc8-c900-4c1a-8c47-4b505370940a","year":2023},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:c4bee805c3d59dc232c828ca36d49d761dff41c64e547a65ded08ecd466f381a","observation_id":"90037464-168e-4e64-8291-c62f5b392ac3","resolution":{"observed_at":"2026-05-13T23:30:00.860438Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.11432","last_updated":"2021-11-22T18:59:55Z","snapshot_observed_at":"2026-07-06T12:11:02.119174Z","submitted_at":"2021-11-22T18:59:55Z","title":"Florence: A New Foundation Model for Computer Vision","version":1},"cited_work":{"arxiv_id":"2111.11432","doi":null,"metadata_source":"pith","pith_arxiv_id":"2111.11432","snapshot_observed_at":"2026-07-04T20:00:08.247992Z","title":"Florence: A New Foundation Model for Computer Vision","venue":"cs.CV","work_id":"99823072-36a8-4b10-9ef5-a7f91da74650","year":2021},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/2111.11432","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:2813fbbd3ed295b0565fe0712147d93632989daf63efcabdbe116a0e92a55bae","observation_id":"33860558-ca6c-4577-a7c3-470807f3a221","resolution":{"observed_at":"2026-05-16T09:38:09.598269Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Merlot reserve: Neural script knowledge through vision and language and sound","venue":null,"work_id":"4f5f02fb-1008-4638-8ecd-2e3301dbf357","year":2022},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:99ec04625c8ef85f1dfcc08fcb494250fcfad44109236b4a0d3dba949e021c26","observation_id":"5345611c-17fc-423d-9288-17a8db0c6650","resolution":{"observed_at":"2026-05-13T23:30:00.836023Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Merlot: Multimodal neural script knowledge models","venue":null,"work_id":"27722bef-dbf8-4fdd-9e59-23d211f9774a","year":2021},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:1f3e5ebd491a14331e1b2f81cd2c45dbba43429eba076d6d7f7fff071b19ff60","observation_id":"e4788a76-ddfa-4a4e-b0ca-ca9660cf8c48","resolution":{"observed_at":"2026-05-13T23:30:00.847015Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.02414","last_updated":"2023-10-25T05:22:43Z","snapshot_observed_at":"2026-08-02T04:26:53.194797Z","submitted_at":"2022-10-05T17:34:44Z","title":"GLM-130B: An Open Bilingual Pre-trained Model","version":2},"cited_work":{"arxiv_id":"2210.02414","doi":"10.48550/arxiv.2210.02414","metadata_source":"pith","pith_arxiv_id":"2210.02414","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GLM-130B: An Open Bilingual Pre-trained Model","venue":"cs.CL","work_id":"b6ef9427-af31-4c7c-8c34-bee621978c23","year":2022},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/2210.02414","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:c3bf9c7fa00c26e20b8ac65d630cfc165f9755a365ca02f127810f5ef22602f7","observation_id":"4ad12f5e-f804-442c-856d-7776cf171457","resolution":{"observed_at":"2026-05-14T17:41:34.452877Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.01068","last_updated":"2022-06-21T17:04:40Z","snapshot_observed_at":"2026-08-06T03:13:37.403059Z","submitted_at":"2022-05-02T17:49:50Z","title":"OPT: Open Pre-trained Transformer Language Models","version":4},"cited_work":{"arxiv_id":"2205.01068","doi":"10.48550/arxiv.2205.01068","metadata_source":"pith","pith_arxiv_id":"2205.01068","snapshot_observed_at":"2026-07-11T03:37:45.880117Z","title":"OPT: Open Pre-trained Transformer Language Models","venue":"cs.CL","work_id":"d7ff3b21-1fff-4cf4-952a-4714e3ef2307","year":2022},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/2205.01068","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:2d41fb64d9b37918256c6153d6fd93b0907508d54abd302e185a783b6955eee0","observation_id":"00743eca-d43d-4f66-8b69-91eb324e821b","resolution":{"observed_at":"2026-05-13T23:30:00.739742Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-25T10:53:17.026227+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-25T10:53:17.026227+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.10592","last_updated":"2023-10-02T16:38:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-20T18:25:35Z","title":"MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models","version":2},"cited_work":{"arxiv_id":"2304.10592","doi":"10.18653/v1/2024.findings-emnlp.692","metadata_source":"pith","pith_arxiv_id":"2304.10592","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models","venue":"cs.CV","work_id":"a7e3a737-e007-42bc-be89-c4d34c5ee071","year":2023},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/2304.10592","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:b096fa5dff6c1245ae28810a7c19663d90f00763231e93303da9a0fdca9f48a4","observation_id":"8a858f91-89d1-4bed-a889-c4f5b6260a7f","resolution":{"observed_at":"2026-05-13T23:30:00.745883Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-23T17:23:55.433296+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T17:23:55.433296+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Describe the following image concisely","venue":null,"work_id":"2fcc0727-9787-46ff-92d4-c647c43af2af","year":2020},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:cd6a218663b42e34b8b2df8e9aef30e4b3bc809c548c78ebfbf25e49baed1430","observation_id":"acf393d8-c28a-4406-8a0d-58d01bf290c3","resolution":{"observed_at":"2026-05-13T23:30:00.870039Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding"},"reference_resolution":{"displayed":59,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":1,"verified_exact":29,"verified_fuzzy":28},"total_outbound_references":59},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 59 of 59 outbound references and 100 inbound Pith citation observations for arXiv:2305.06355."}