{"as_of":"2026-08-11T18:58:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:92e95486b3209d341b091763c9bc5ad27c79a0c3e212d8bfa7905b545e9af074","coverage":[{"denominator":76,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":76,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-18T00:30:12.190492Z","state":"measured"},{"denominator":156,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":156,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":80,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":80,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T16:58:29.713870Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":2,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2406.04264","last_updated":"2025-01-01T15:53:58Z","snapshot_observed_at":"2026-08-03T20:38:36.602554Z","submitted_at":"2024-06-06T17:09:32Z","title":"MLVU: Benchmarking Multi-task Long Video Understanding","version":3},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-14T19:55:26.333923Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2406.04264"},"observation_digest":"sha256:a7b5cf5d3aa9ad2d8d2d0adff0d989512192858fd1e2f5461d4354e7923acfb1","observation_id":"5ab2c9a1-5bd2-4dca-8ab2-35b1de4d3d59","resolution":{"observed_at":"2026-05-18T00:30:12.519784Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2406.08035","last_updated":"2025-08-09T10:54:59Z","snapshot_observed_at":"2026-08-08T19:38:26.415599Z","submitted_at":"2024-06-12T09:36:52Z","title":"LVBench: An Extreme Long Video Understanding Benchmark","version":3},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-19T11:55:30.048525Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2406.08035"},"observation_digest":"sha256:87f6a38a75496209d56649e1c4f1dddd08eda3d76832e0bd5896ae2b1a9312ac","observation_id":"0776a99f-8bb0-48c6-afe1-d07ef7994b04","resolution":{"observed_at":"2026-05-19T11:55:30.123446Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":253,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:4020a22400ca9a8e33bf11682a928d9cc548aa4f69224a24690039b72854eb38","observation_id":"1ec9fb76-6e93-432d-bbe2-38363f906f95","resolution":{"observed_at":"2026-05-20T06:20:36.531913Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2408.10188","last_updated":"2024-12-13T02:32:06Z","snapshot_observed_at":"2026-08-05T14:57:53.592979Z","submitted_at":"2024-08-19T17:48:08Z","title":"LongVILA: Scaling Long-Context Visual Language Models for Long Videos","version":6},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-17T03:51:25.396887Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2408.10188"},"observation_digest":"sha256:aff52e8ac81642b08358e1558b6a8ccecefc14ad3d233f40e94a8437abbcceb1","observation_id":"8f25afe6-c526-4608-ba71-a17de0a9feac","resolution":{"observed_at":"2026-05-18T00:30:12.519784Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-05-10T23:20:32.330351Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2410.02713"},"observation_digest":"sha256:423a702d87bf9a147693a0ed7da4f0b65cddb08b40c327a319c504a1be5b63cf","observation_id":"13c7293b-8856-4dbe-af2f-33d45d80e6b2","resolution":{"observed_at":"2026-05-18T00:30:12.519784Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2412.05271","last_updated":"2025-09-26T12:52:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-06T18:57:08Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","version":5},"reference_index":263,"source":"pdf_text","source_observed_at":"2026-05-10T13:23:57.588851Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2412.05271"},"observation_digest":"sha256:53afe239d55c1fda33883c3cdda111379794c845765280d4b239510f570b4f43","observation_id":"7572919b-ecc4-47ab-ba8a-5423514ad5e5","resolution":{"observed_at":"2026-05-18T00:30:12.519784Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-11T16:58:29.713870Z","title":"Longvideobench: A benchmark for long-context interleaved video-language understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.09582","last_updated":"2025-01-18T00:52:42Z","snapshot_observed_at":"2026-08-11T16:51:47.784125Z","submitted_at":"2024-12-12T18:54:48Z","title":"Neptune: The Long Orbit to Benchmarking Long Video Understanding","version":2},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-11T16:58:29.713870Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2412.09582"},"observation_digest":"sha256:65a3cb5c1719ad1af6ac86bee8cc2064f032899da539ad32a7b623cfe975922b","observation_id":"c054daf9-5de4-4596-8ef1-d3b6a04e9b40","resolution":{"observed_at":"2026-08-11T16:58:29.713870Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-11T16:56:42.699473Z","title":"Longvideobench: A benchmark for long-context inter- leaved video-language understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.09613","last_updated":"2024-12-12T18:59:40Z","snapshot_observed_at":"2026-08-11T17:35:07.633589Z","submitted_at":"2024-12-12T18:59:40Z","title":"PVC: Progressive Visual Token Compression for Unified Image and Video Processing in Large Vision-Language Models","version":1},"reference_index":131,"source":"pdf_text","source_observed_at":"2026-08-11T16:56:42.699473Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2412.09613"},"observation_digest":"sha256:09b8f5c82107e768cc3b6fc4023f040826bd61893aecf35b27a56fab20221912","observation_id":"5b7e8238-17d8-484a-b613-4a2af470a1d6","resolution":{"observed_at":"2026-08-11T16:56:42.699473Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-11T16:11:10.676719Z","title":"Longvideobench: A benchmark for long-context interleaved video-language understanding, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10360","last_updated":"2024-12-13T18:53:24Z","snapshot_observed_at":"2026-08-11T17:37:01.087489Z","submitted_at":"2024-12-13T18:53:24Z","title":"Apollo: An Exploration of Video Understanding in Large Multimodal Models","version":1},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-11T16:11:10.676719Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2412.10360"},"observation_digest":"sha256:ad415b4f775ff077a466e092f7366f61a301fd0d2e848c5dfdb7cb73f485f49d","observation_id":"532c93ea-2b2f-4399-be6b-79ca608ecb77","resolution":{"observed_at":"2026-08-11T16:11:10.676719Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-11T13:52:57.176808Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.12735","last_updated":"2024-12-17T09:57:21Z","snapshot_observed_at":"2026-08-11T17:35:17.874397Z","submitted_at":"2024-12-17T09:57:21Z","title":"GIRAFFE: Design Choices for Extending the Context Length of Visual Language Models","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-11T13:52:57.176808Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2412.12735"},"observation_digest":"sha256:64fee2e0145da090726fd954c072e1bf32363be126ca8a77f21c38ba306bf5ac","observation_id":"1d1af7f1-7ea6-4bf5-b7af-944bef1cc46d","resolution":{"observed_at":"2026-08-11T13:52:57.176808Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-11T13:44:50.497730Z","title":"Longvideobench: A benchmark for long-context inter- leaved video-language understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.12833","last_updated":"2025-05-28T09:22:24Z","snapshot_observed_at":"2026-08-11T17:45:15.774121Z","submitted_at":"2024-12-17T11:54:47Z","title":"FocusChat: Text-guided Long Video Understanding via Spatiotemporal Information Filtering","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-11T13:44:50.497730Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2412.12833"},"observation_digest":"sha256:153c5d9f83b2ed1a3072151c2bbdedca83a96366aba7daca386c634729507100","observation_id":"168359ec-92d4-4ff7-be2d-8a08834f87b4","resolution":{"observed_at":"2026-08-11T13:44:50.497730Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-11T00:38:50.106807Z","title":"Available: https://arxiv.org/abs/2407.15754","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.19442","last_updated":"2025-07-30T05:24:46Z","snapshot_observed_at":"2026-08-11T00:33:34.388194Z","submitted_at":"2024-12-27T04:17:57Z","title":"A Survey on Large Language Model Acceleration based on KV Cache Management","version":3},"reference_index":300,"source":"pdf_text","source_observed_at":"2026-08-11T00:38:50.106807Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2412.19442"},"observation_digest":"sha256:6dddfe7257a5d520194c80c1daf717cd21491130cd69b58936cc6ee112828f33","observation_id":"ce983033-dab8-43e7-aea3-502ceda644f5","resolution":{"observed_at":"2026-08-11T00:38:50.106807Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2501.00574","last_updated":"2025-07-13T16:21:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-31T18:01:23Z","title":"VideoChat-Flash: Hierarchical Compression for Long-Context Video Modeling","version":4},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-18T04:02:43.261543Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2501.00574"},"observation_digest":"sha256:101d23e36747a232b5e5ac140021c0f6e2e792721261cb5a17cf0f968b93c788","observation_id":"4de7f155-5d02-4109-9a9a-0c302c4cdc64","resolution":{"observed_at":"2026-05-18T04:02:43.332710Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2501.02955","last_updated":"2026-05-12T15:02:48Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-06T11:57:38Z","title":"MotionBench: Benchmarking and Improving Fine-grained Video Motion Understanding for Vision Language Models","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-23T05:44:31.546843Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2501.02955"},"observation_digest":"sha256:0bddf4e6f516f7c9262ca9810d462d6d955ef2a9e44a499c7bff2dfa6fe493d4","observation_id":"cc24b07d-9249-4d58-9b55-3dc15e5479ec","resolution":{"observed_at":"2026-05-23T05:45:28.327036Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:a24d7fa807b262646c10badef5cb4686c62f38fa8777c15222b0090f63ebf24a","observation_id":"56b398d2-a2a2-4efb-9779-f18fdb255c2f","resolution":{"observed_at":"2026-05-18T00:30:12.519784Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2501.13826","last_updated":"2025-01-23T16:51:47Z","snapshot_observed_at":"2026-07-06T20:25:03.950783Z","submitted_at":"2025-01-23T16:51:47Z","title":"Video-MMMU: Evaluating Knowledge Acquisition from Multi-Discipline Professional Videos","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-14T00:32:41.059558Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2501.13826"},"observation_digest":"sha256:35e5295cb67d216098b9d8ee1dcd6bb2f9b5b441a55c8aaefd47be8971011fc8","observation_id":"73d0fb67-fa4f-4517-ba83-ea220dd15072","resolution":{"observed_at":"2026-05-18T00:30:12.519784Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-10T15:35:30.270844Z","title":"Longvideobench: A benchmark for long-context interleaved video-language understanding, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13919","last_updated":"2025-09-01T07:50:58Z","snapshot_observed_at":"2026-08-11T01:32:58.320917Z","submitted_at":"2025-01-23T18:58:03Z","title":"Temporal Preference Optimization for Long-Form Video Understanding","version":3},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-10T15:35:30.270844Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2501.13919"},"observation_digest":"sha256:a0fa10aa760392422caddbe82fbffe9b3e965547f18e7c374c047022edb48812","observation_id":"41e79b09-b4b4-4c5d-a509-1a2bda6a9979","resolution":{"observed_at":"2026-08-10T15:35:30.270844Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-10T14:18:17.964511Z","title":"Longvideobench: A benchmark for long-context interleaved video-language understanding.arXiv preprint arXiv:2407.15754, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.15513","last_updated":"2025-06-10T14:30:19Z","snapshot_observed_at":"2026-08-10T14:10:29.461568Z","submitted_at":"2025-01-26T13:10:12Z","title":"TinyLLaVA-Video: Towards Smaller LMMs for Video Understanding with Group Resampler","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-10T14:18:17.964511Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2501.15513"},"observation_digest":"sha256:97cb864b0ec62ee205a718c3582148d0cb7f5e6ec7a9ace3af34aa8c6f5d3573","observation_id":"d49048db-9381-46d5-b43a-c05e9d6b6cf5","resolution":{"observed_at":"2026-08-10T14:18:17.964511Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-09T15:04:36.978528Z","title":"Longvideobench: A benchmark for long-context interleaved video-language understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01549","last_updated":"2025-02-03T17:30:19Z","snapshot_observed_at":"2026-08-09T16:01:36.816495Z","submitted_at":"2025-02-03T17:30:19Z","title":"VideoRAG: Retrieval-Augmented Generation with Extreme Long-Context Videos","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-09T15:04:36.978528Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2502.01549"},"observation_digest":"sha256:11f9bf2954d7551fb854b1cf4b51edd0e052b46eaf84b6227148e47c52063fa2","observation_id":"c31406fa-23c2-4c2a-bcd2-cd967d7b8ef5","resolution":{"observed_at":"2026-08-09T15:04:36.978528Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-08T22:47:39.326631Z","title":"Longvideobench: A benchmark for long-context inter- leaved video-language understanding.arXiv preprint arXiv:2407.15754, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.04328","last_updated":"2025-06-02T19:33:24Z","snapshot_observed_at":"2026-08-10T02:46:21.304913Z","submitted_at":"2025-02-06T18:59:55Z","title":"Ola: Pushing the Frontiers of Omni-Modal Language Model","version":3},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-08T22:47:39.326631Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2502.04328"},"observation_digest":"sha256:fa756b440a5d0522e0e3f515210249a0deabdcb03a03802f5ba4279a3dfcf4a9","observation_id":"21dda9e6-26c1-4c46-ab0d-76e8d652c64f","resolution":{"observed_at":"2026-08-08T22:47:39.326631Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-23T02:25:04.405036Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2502.13923"},"observation_digest":"sha256:0f4337d299c7f586f0c60aaa53dff4079ff6c9c24275871f34dc30cd4f1dfc1d","observation_id":"c74e2860-64ee-48c9-aeeb-b2636d4cfa30","resolution":{"observed_at":"2026-05-23T02:25:18.985906Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2504.10479","last_updated":"2025-04-19T03:47:21Z","snapshot_observed_at":"2026-08-10T18:37:57.419939Z","submitted_at":"2025-04-14T17:59:25Z","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","version":3},"reference_index":130,"source":"pdf_text","source_observed_at":"2026-05-10T13:41:07.991012Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2504.10479"},"observation_digest":"sha256:f9ec55b155f6c051348bb3ef6a722cc20c4ad51a8d18217b92e22b95f1d5e6da","observation_id":"63a6737f-eebb-4000-9f4b-f42660ad1d14","resolution":{"observed_at":"2026-05-18T00:30:12.519784Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2505.15269","last_updated":"2026-04-23T12:54:38Z","snapshot_observed_at":"2026-08-10T22:50:24.267781Z","submitted_at":"2025-05-21T08:47:15Z","title":"LiveVLM: Efficient Online Video Understanding via Streaming-Oriented KV Cache and Retrieval","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-22T14:26:59.015559Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2505.15269"},"observation_digest":"sha256:80a25fe7b7e785495a5637aec049b7d987ba269169dedc154f3f96241f477d83","observation_id":"8043b115-0ff2-4797-8d51-347d8df59fdd","resolution":{"observed_at":"2026-05-22T14:31:40.737174Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-07T15:09:01.557498Z","title":"Longvideobench: A benchmark for long-context interleaved video-language understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16175","last_updated":"2025-05-31T13:43:36Z","snapshot_observed_at":"2026-08-08T10:58:05.073646Z","submitted_at":"2025-05-22T03:26:50Z","title":"QuickVideo: Real-Time Long Video Understanding with System Algorithm Co-Design","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T15:09:01.557498Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2505.16175"},"observation_digest":"sha256:645be1a57bb5149172b6c77d53d3232ecd45bfd440fe4ad5c3e32ec1c37fa46e","observation_id":"12eb9dac-3327-471b-a1bd-e057e52bebb7","resolution":{"observed_at":"2026-08-07T15:09:01.557498Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-07T13:32:33.762529Z","title":"Longvideobench: A benchmark for long-context interleaved video-language understanding, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21487","last_updated":"2025-05-27T17:54:07Z","snapshot_observed_at":"2026-08-07T13:24:37.243429Z","submitted_at":"2025-05-27T17:54:07Z","title":"Hardware-Efficient Attention for Fast Decoding","version":1},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-08-07T13:32:33.762529Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2505.21487"},"observation_digest":"sha256:8d154f137d1775fdaa5df3ff4119db9b50f1bdbad9bc9feb611f579b2859c296","observation_id":"692ab0ed-be0c-493f-b724-84e4e1f72e74","resolution":{"observed_at":"2026-08-07T13:32:33.762529Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-07T12:48:04.238595Z","title":"Longvideobench: A benchmark for long-context interleaved video-language understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23484","last_updated":"2025-05-29T14:34:25Z","snapshot_observed_at":"2026-08-09T05:32:01.732887Z","submitted_at":"2025-05-29T14:34:25Z","title":"VCapsBench: A Large-scale Fine-grained Benchmark for Video Caption Quality Evaluation","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:04.238595Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2505.23484"},"observation_digest":"sha256:e9e570afa7f19c373e75951c7ea7da47562fda18ba61ca8073605708e6697f27","observation_id":"91e2bce3-dc08-4307-8092-8bb49921bed9","resolution":{"observed_at":"2026-08-07T12:48:04.238595Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-07T12:40:46.344645Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23922","last_updated":"2025-05-29T18:15:07Z","snapshot_observed_at":"2026-08-09T06:58:13.030008Z","submitted_at":"2025-05-29T18:15:07Z","title":"ScaleLong: A Multi-Timescale Benchmark for Long Video Understanding","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-07T12:40:46.344645Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2505.23922"},"observation_digest":"sha256:75ecc192bf1d3bb99a69f090ffaeb637bea04b147cc813dfe9d46b7a23b39a2e","observation_id":"a2bdbbba-6147-4a80-b3e4-0fcdc8e64677","resolution":{"observed_at":"2026-08-07T12:40:46.344645Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-07T11:59:09.727032Z","title":"Longvideobench: A benchmark for long- context interleaved video-language understanding, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.00993","last_updated":"2025-06-01T12:49:39Z","snapshot_observed_at":"2026-08-08T18:44:33.509277Z","submitted_at":"2025-06-01T12:49:39Z","title":"FlexSelect: Flexible Token Selection for Efficient Long Video Understanding","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:09.727032Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2506.00993"},"observation_digest":"sha256:de20268ec46a56727cbc5d3f4ee812c00be0be44f6b3c387ef949403ea7ca519","observation_id":"07b2a234-bd0f-4ea4-832d-8429a0bf96d2","resolution":{"observed_at":"2026-08-07T11:59:09.727032Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-07T11:51:29.369665Z","title":"Longvideobench: A benchmark for long-context interleaved video-language understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01274","last_updated":"2026-06-11T17:06:50Z","snapshot_observed_at":"2026-08-10T04:03:56.011290Z","submitted_at":"2025-06-02T03:08:07Z","title":"ReFoCUS: Reinforcement-guided Frame Optimization for Contextual Understanding","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:29.369665Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2506.01274"},"observation_digest":"sha256:25df3957f61ab5b40f57a16429f6b891a393d1f0106ee624d45ab3e26a065f36","observation_id":"ff2fa1cb-25de-4f53-b015-109f3cbeb062","resolution":{"observed_at":"2026-08-07T11:51:29.369665Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-07T10:56:05.383591Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.03990","last_updated":"2025-06-04T14:17:42Z","snapshot_observed_at":"2026-08-09T05:23:11.832180Z","submitted_at":"2025-06-04T14:17:42Z","title":"DynTok: Dynamic Compression of Visual Tokens for Efficient and Effective Video Understanding","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-07T10:56:05.383591Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2506.03990"},"observation_digest":"sha256:8ad666023ffa908097a24e28ce804fd4f9af166523641db3025339f574121105","observation_id":"ca11976d-f9ad-4707-a57a-bbee01e75148","resolution":{"observed_at":"2026-08-07T10:56:05.383591Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-07T10:26:32.111857Z","title":"Jimmy Wu, Rika Antonova, Adam Kan, Marion Lepert, Andy Zeng, Shuran Song, Jeannette Bohg, Szymon Rusinkiewicz, and Thomas Funkhouser","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05523","last_updated":"2025-06-05T19:12:45Z","snapshot_observed_at":"2026-08-07T22:11:58.027730Z","submitted_at":"2025-06-05T19:12:45Z","title":"MORSE-500: A Programmatically Controllable Video Benchmark to Stress-Test Multimodal Reasoning","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T10:26:32.111857Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2506.05523"},"observation_digest":"sha256:642201b4ab9d499d30c614535ceadd010de93b868f2737351c77743867163e0e","observation_id":"97733af1-ffa8-44d9-afbc-b1ffe073f836","resolution":{"observed_at":"2026-08-07T10:26:32.111857Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-06T23:13:11.123012Z","title":"Longvideobench: A benchmark for long-context interleaved video-language understanding.arXiv preprint arXiv:2407.15754, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.19225","last_updated":"2025-06-24T01:19:56Z","snapshot_observed_at":"2026-08-07T21:07:46.308380Z","submitted_at":"2025-06-24T01:19:56Z","title":"Video-XL-2: Towards Very Long-Video Understanding Through Task-Aware KV Sparsification","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T23:13:11.123012Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2506.19225"},"observation_digest":"sha256:9849c33212ca2aaa5ae5c8c3e93d0f0cfeb9e920dbb78ca83e83395174a29309","observation_id":"21397d08-1c0d-4164-9c4f-9539c132815f","resolution":{"observed_at":"2026-08-06T23:13:11.123012Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-06T22:36:39.027331Z","title":"Longvideobench: A benchmark for long-context interleaved video-language understanding.arXiv preprint arXiv:2407.15754, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.21184","last_updated":"2025-06-26T12:43:43Z","snapshot_observed_at":"2026-08-08T13:40:41.778180Z","submitted_at":"2025-06-26T12:43:43Z","title":"Task-Aware KV Compression For Cost-Effective Long Video Understanding","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T22:36:39.027331Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2506.21184"},"observation_digest":"sha256:ef759af5b6c69ee08e960f7794a503300b2394c745aca4767ea562292f3746b4","observation_id":"dc1e9d7c-528b-49e6-9fda-5a5a44fc868c","resolution":{"observed_at":"2026-08-06T22:36:39.027331Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-06T20:29:55.224420Z","title":"Longvideobench: A benchmark for long-context inter- leaved video-language understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02591","last_updated":"2025-07-23T07:25:27Z","snapshot_observed_at":"2026-08-09T23:16:14.999128Z","submitted_at":"2025-07-03T12:55:16Z","title":"AuroraLong: Bringing RNNs Back to Efficient Open-Ended Video Understanding","version":3},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-06T20:29:55.224420Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2507.02591"},"observation_digest":"sha256:a1350e500c4b756a9a9591245ac46205e35fa48354a5476b140cf947fad4d555","observation_id":"1e423a24-2f0a-46a3-8cf4-c44392ca6812","resolution":{"observed_at":"2026-08-06T20:29:55.224420Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-06T23:12:00.487576Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02904","last_updated":"2025-06-24T06:08:35Z","snapshot_observed_at":"2026-08-10T17:23:59.596263Z","submitted_at":"2025-06-24T06:08:35Z","title":"Enhancing Sports Strategy with Video Analytics and Data Mining: Assessing the effectiveness of Multimodal LLMs in tennis video analysis","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-06T23:12:00.487576Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2507.02904"},"observation_digest":"sha256:f6ece3f4be5ea94dcd5fe1eb331349e8feb9d63622ae064df62200d36ced2c5d","observation_id":"5f8e6737-4707-4583-9059-de5e759740c3","resolution":{"observed_at":"2026-08-06T23:12:00.487576Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-06T18:10:14.340618Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09068","last_updated":"2025-07-23T13:06:44Z","snapshot_observed_at":"2026-08-11T10:43:34.020108Z","submitted_at":"2025-07-11T23:07:04Z","title":"Infinite Video Understanding","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T18:10:14.340618Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2507.09068"},"observation_digest":"sha256:a254a1e2f1821a70f68c2ff89c6be1e32d77061adf322f939ddf3d1de4286e1f","observation_id":"b6143b35-bfaa-484d-9084-4430ec886d4b","resolution":{"observed_at":"2026-08-06T18:10:14.340618Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-06T15:26:02.297110Z","title":"Longvideobench: A benchmark for long-context interleaved video-language understanding, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.16863","last_updated":"2026-06-24T02:13:52Z","snapshot_observed_at":"2026-08-09T02:01:23.087988Z","submitted_at":"2025-07-21T21:50:16Z","title":"Position: Reasoning After Perception Means Reasoning Without Vision","version":2},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-06T15:26:02.297110Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2507.16863"},"observation_digest":"sha256:490b4036a312d6b283e4d6a224294ef88c840d4d995928342fa8ae84fe3462fe","observation_id":"a1b6f5b0-a2c5-4240-9fbc-2148c901f64b","resolution":{"observed_at":"2026-08-06T15:26:02.297110Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-06T04:49:41.776592Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.03039","last_updated":"2025-08-05T03:33:24Z","snapshot_observed_at":"2026-08-07T14:34:47.687957Z","submitted_at":"2025-08-05T03:33:24Z","title":"VideoForest: Person-Anchored Hierarchical Reasoning for Cross-Video Question Answering","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T04:49:41.776592Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2508.03039"},"observation_digest":"sha256:f732a2c0718ddbb9d5e9ea8267e36185b35d388029b548b11aad49a5564dd247","observation_id":"28fd947c-bfe8-4d0b-8954-e19b2f9e491f","resolution":{"observed_at":"2026-08-06T04:49:41.776592Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2508.18265","last_updated":"2025-08-27T14:39:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-25T17:58:17Z","title":"InternVL3.5: Advancing Open-Source Multimodal Models in Versatility, Reasoning, and Efficiency","version":2},"reference_index":150,"source":"pdf_text","source_observed_at":"2026-05-10T11:58:58.660564Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2508.18265"},"observation_digest":"sha256:3808691b02692781940d9ac785b9fd73e69299a39085f6419a241cd1f11a3060","observation_id":"c08bfd29-d16f-4ec2-95f4-4d6babff51e8","resolution":{"observed_at":"2026-05-18T00:30:12.519784Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T11:56:37.767697Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.02164","last_updated":"2025-09-02T10:14:55Z","snapshot_observed_at":"2026-08-07T04:11:09.551801Z","submitted_at":"2025-09-02T10:14:55Z","title":"Omnidirectional Spatial Modeling from Correlated Panoramas","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-05T11:56:37.767697Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2509.02164"},"observation_digest":"sha256:31c4b375634ca6b8ef4bd55dd5d22ca3839dba14e3879bf33317fda6a843083f","observation_id":"d2a7d53a-d7c2-465b-b727-eab4539bc83b","resolution":{"observed_at":"2026-08-05T11:56:37.767697Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2512.08410","last_updated":"2026-04-09T06:44:08Z","snapshot_observed_at":"2026-08-10T21:26:08.856126Z","submitted_at":"2025-12-09T09:40:20Z","title":"Towards Effective Long Video Understanding of Multimodal Large Language Models via One-shot Clip Retrieval","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-17T00:10:01.596422Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2512.08410"},"observation_digest":"sha256:d1904d821427c9beb77200af0b4153edd01f416a804f68227e8d5f3da2101a39","observation_id":"30f607a5-3d78-4ca0-bd7e-55cf4a7b3c29","resolution":{"observed_at":"2026-05-18T00:30:12.519784Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-03T15:26:15.333859Z","title":"Longvideobench: A benchmark for long-context inter- leaved video-language understanding.arXiv preprint arXiv:2407.15754, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2512.16978","last_updated":"2026-06-16T16:38:58Z","snapshot_observed_at":"2026-08-07T04:10:11.766593Z","submitted_at":"2025-12-18T18:59:27Z","title":"A Benchmark for Omni-Modal Reasoning in Long Videos","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-03T15:26:15.333859Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2512.16978"},"observation_digest":"sha256:29e84060e4cc2fee7b85b0bf15137c3faf0ed3ae0d2a4ee94c1be5cd20125d26","observation_id":"568f6f02-179f-4512-827a-9618bc607977","resolution":{"observed_at":"2026-08-03T15:26:15.333859Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2602.02276","last_updated":"2026-02-02T16:17:38Z","snapshot_observed_at":"2026-07-06T22:44:09.804048Z","submitted_at":"2026-02-02T16:17:38Z","title":"Kimi K2.5: Visual Agentic Intelligence","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-05-10T16:09:05.225767Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2602.02276"},"observation_digest":"sha256:aab7cb67760f5db03742c6b8a4887cf8cfa179fb75b60a7e6b69b8076751409e","observation_id":"24ed4400-a679-49f9-a356-461ff8d083e8","resolution":{"observed_at":"2026-05-18T00:30:12.519784Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-03T03:22:53.771827Z","title":"Longvideobench: A benchmark for long-context interleaved video-language understanding, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.08426","last_updated":"2026-05-25T11:18:29Z","snapshot_observed_at":"2026-08-03T03:22:52.672693Z","submitted_at":"2026-02-09T09:31:06Z","title":"Prism: Spectral-Aware Block-Sparse Attention","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-03T03:22:53.771827Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2602.08426"},"observation_digest":"sha256:87286f2e22d6209b061fef7537efc9a22fbc8f0142381a32ed9c6bc35e2d3baf","observation_id":"a58d0407-386b-4526-9330-34952f5f1076","resolution":{"observed_at":"2026-08-03T03:22:53.771827Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2603.22911","last_updated":"2026-04-12T14:01:49Z","snapshot_observed_at":"2026-08-11T18:11:42.555848Z","submitted_at":"2026-03-24T08:01:16Z","title":"ForestPrune: High-ratio Visual Token Compression for Video Multimodal Large Language Models via Spatial-Temporal Forest Modeling","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-15T00:56:47.841355Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2603.22911"},"observation_digest":"sha256:9614241c90247fe0d9f93a7c90cbac8128c2decb48ed029dbdb4504a2a4c16f7","observation_id":"a616e393-cdc3-4dc4-bb3c-790457920b10","resolution":{"observed_at":"2026-05-18T00:30:12.519784Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2604.08703","last_updated":"2026-04-09T18:51:16Z","snapshot_observed_at":"2026-07-06T22:57:45.084987Z","submitted_at":"2026-04-09T18:51:16Z","title":"QoS-QoE Translation with Large Language Model","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-10T17:01:57.703684Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2604.08703"},"observation_digest":"sha256:3f8513c05ec5bcb96ed0120a88870a21de5da4c0978901b692ea590ef0e3d03d","observation_id":"91433a77-3929-4f0a-9cb5-d7d6801a2fca","resolution":{"observed_at":"2026-05-18T00:30:12.519784Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2604.10060","last_updated":"2026-04-11T06:54:56Z","snapshot_observed_at":"2026-08-10T21:22:44.566409Z","submitted_at":"2026-04-11T06:54:56Z","title":"Mosaic: Cross-Modal Clustering for Efficient Video Understanding","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-10T16:07:36.133404Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2604.10060"},"observation_digest":"sha256:8cbfffafc0482b177326708b326a8bf29bcab3bb987ea328616c51ba135cf8c7","observation_id":"a1d6352e-cdb3-4735-b732-b06e87014c97","resolution":{"observed_at":"2026-05-18T00:30:12.519784Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2604.16893","last_updated":"2026-04-18T07:56:32Z","snapshot_observed_at":"2026-07-06T23:04:05.813982Z","submitted_at":"2026-04-18T07:56:32Z","title":"EasyVideoR1: Easier RL for Video Understanding","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-10T07:41:27.231098Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2604.16893"},"observation_digest":"sha256:46d5fd9f5b025b6866957a743257d66448062af0fcad40e0bf1b4d2d3096da13","observation_id":"1e17bb62-8cc7-4cd1-a6af-9a6a311f4557","resolution":{"observed_at":"2026-05-18T00:30:12.519784Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2605.05225","last_updated":"2026-06-05T13:47:55Z","snapshot_observed_at":"2026-08-11T18:46:22.698604Z","submitted_at":"2026-04-19T07:25:39Z","title":"MACS: Modality-Aware Capacity Scaling for Efficient Multimodal MoE Inference","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-10T06:47:25.155437Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2605.05225"},"observation_digest":"sha256:411a2132479769730ac2fa9a93adf06150aea88d7272e8b18d58aae072bb8f4d","observation_id":"c3f8103a-a35a-47ef-84f1-db1a2aa635fa","resolution":{"observed_at":"2026-05-18T00:30:12.519784Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2605.05225","last_updated":"2026-06-05T13:47:55Z","snapshot_observed_at":"2026-08-11T18:46:22.698604Z","submitted_at":"2026-04-19T07:25:39Z","title":"MACS: Modality-Aware Capacity Scaling for Efficient Multimodal MoE Inference","version":2},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-05-11T01:29:26.298131Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2605.05225"},"observation_digest":"sha256:cb6796fd95411ead1629f0fc5a1ab8b7f4bd8837870ff5933f194490e796fc3a","observation_id":"219e4f1a-bdcf-42fb-9c87-0463583378fa","resolution":{"observed_at":"2026-05-18T00:30:12.519784Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2605.07355","last_updated":"2026-05-08T07:08:54Z","snapshot_observed_at":"2026-08-10T21:26:10.490587Z","submitted_at":"2026-05-08T07:08:54Z","title":"TTF: Temporal Token Fusion for Efficient Video-Language Model","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-05-11T01:55:32.319344Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2605.07355"},"observation_digest":"sha256:f1568f6a7831f85f8ba4d609d793f3640e4bc06c85ac30e7a102b6e144006b06","observation_id":"39ca4d1d-1ee5-425e-9494-154691c33459","resolution":{"observed_at":"2026-05-18T00:30:12.519784Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2605.14906","last_updated":"2026-05-14T14:41:17Z","snapshot_observed_at":"2026-08-02T09:41:14.453439Z","submitted_at":"2026-05-14T14:41:17Z","title":"MemLens: Benchmarking Multimodal Long-Term Memory in Large Vision-Language Models","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-06-30T21:00:25.664841Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2605.14906"},"observation_digest":"sha256:ba35d2299bae9839c42ee200fe6b640cb8a2dc277db7af1f63b9fd4025aebe19","observation_id":"297fefa3-981e-452f-9939-0353cd9d22dd","resolution":{"observed_at":"2026-06-30T21:05:04.222455Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2605.17921","last_updated":"2026-06-01T06:29:58Z","snapshot_observed_at":"2026-07-06T23:28:50.117638Z","submitted_at":"2026-05-18T06:29:44Z","title":"An Efficient Streaming Video Understanding Framework with Agentic Control","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-05-20T11:30:22.151045Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2605.17921"},"observation_digest":"sha256:b9d9b0ddfd38a7849b8e7178d0b44712c73118ec403431e3b9914f6a4cca0022","observation_id":"9f4fb304-530a-4408-8c41-a9f529491be4","resolution":{"observed_at":"2026-05-20T11:33:14.467066Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2605.17921","last_updated":"2026-06-01T06:29:58Z","snapshot_observed_at":"2026-07-06T23:28:50.117638Z","submitted_at":"2026-05-18T06:29:44Z","title":"An Efficient Streaming Video Understanding Framework with Agentic Control","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-30T18:59:35.115402Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2605.17921"},"observation_digest":"sha256:e4e59e346784c7a5c7f2cbd1be2d98e1cc56f3f6fe482e49e1471714df66fdc2","observation_id":"946d8254-cbf5-4c78-bacd-17de86bbea5b","resolution":{"observed_at":"2026-06-30T19:05:00.989979Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2605.19559","last_updated":"2026-05-19T09:02:20Z","snapshot_observed_at":"2026-08-10T16:22:29.533323Z","submitted_at":"2026-05-19T09:02:20Z","title":"EgoCoT-Bench: Benchmarking Grounded and Verifiable Operation-Centric Chain of Thought Reasoning for MLLMs","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-20T05:53:05.450946Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2605.19559"},"observation_digest":"sha256:9f30fa6b6fcb337ee59bca8ae26a2e1d4de1ec985ab714da80cc84c57d0819c2","observation_id":"7d62b82d-39d1-4342-8c30-8ae91ab1bef4","resolution":{"observed_at":"2026-05-20T05:53:22.297930Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2605.30673","last_updated":"2026-07-06T14:01:36Z","snapshot_observed_at":"2026-08-02T07:17:06.461529Z","submitted_at":"2026-05-29T00:06:54Z","title":"TeachObs: A Human-Validated Benchmark for Multimodal Teaching Observation and Model Evaluation","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-28T23:11:43.712391Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2605.30673"},"observation_digest":"sha256:5a5f33f89bdca24d6e010cc32a5dd60016df8ac9a7544624524c782011bcaa80","observation_id":"872929e2-6795-449d-b0d1-de25256d8cfe","resolution":{"observed_at":"2026-06-28T23:12:46.656870Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-07-12T15:34:37.002001Z","title":"LongVideoBench: A Bench- mark for Long-context Interleaved Video-Language Understanding, July 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2605.31027","last_updated":"2026-07-11T04:14:11Z","snapshot_observed_at":"2026-08-02T10:01:28.608360Z","submitted_at":"2026-05-29T08:58:39Z","title":"Multi-Scale Separable Fourier Neural Networks for Solving High-Frequency PDEs","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-07-12T15:34:37.002001Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2605.31027"},"observation_digest":"sha256:b6b5b396b21a939a8ec3bccfd29be02c4b7b3bf4681f40e6d64fd3bbcecf063b","observation_id":"c6244c15-7190-4cf5-a795-16aaeca3cc3a","resolution":{"observed_at":"2026-07-12T15:34:37.002001Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-07-14T18:39:24.915547Z","title":"LongVideoBench: A Bench- mark for Long-context Interleaved Video-Language Understanding, July 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2605.31027","last_updated":"2026-07-11T04:14:11Z","snapshot_observed_at":"2026-08-02T10:01:28.608360Z","submitted_at":"2026-05-29T08:58:39Z","title":"Multi-Scale Separable Fourier Neural Networks for Solving High-Frequency PDEs","version":4},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-07-14T18:39:24.915547Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2605.31027"},"observation_digest":"sha256:e6e66adc1b35d7a02d7322aca6c66dde4d324b7378d35a13b831971817d02cd1","observation_id":"2d42dc1e-3213-40cb-9ef7-06b95c3e94f2","resolution":{"observed_at":"2026-07-14T18:39:24.915547Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2605.31029","last_updated":"2026-05-29T09:01:56Z","snapshot_observed_at":"2026-07-06T23:40:12.939143Z","submitted_at":"2026-05-29T09:01:56Z","title":"PEEK: Picking Essential frames via Efficient Knowledge distillation","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-06-28T22:46:53.704892Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2605.31029"},"observation_digest":"sha256:3e0d2d2bad891cd2b5e16601ecf743980a04f51503910f66196c8d4b7a19d801","observation_id":"55dfa26a-804a-46b5-a080-6a6b4f15a3f5","resolution":{"observed_at":"2026-07-01T19:16:01.311864Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2605.31457","last_updated":"2026-05-29T15:51:12Z","snapshot_observed_at":"2026-08-05T20:26:12.081479Z","submitted_at":"2026-05-29T15:51:12Z","title":"VisionPulse: Dynamic Visual Sparsity for Efficient Multimodal Reasoning","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-28T23:13:15.593994Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2605.31457"},"observation_digest":"sha256:f639e8a2ad905973fb784a72972280bc6703ee724e7b11e70d62797b0c604045","observation_id":"6fd908c0-611d-4160-adf7-a299b677d698","resolution":{"observed_at":"2026-06-29T00:12:50.596604Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2606.07639","last_updated":"2026-06-01T09:07:15Z","snapshot_observed_at":"2026-08-11T17:00:51.145064Z","submitted_at":"2026-06-01T09:07:15Z","title":"MOSS-Video-Preview: Toward Real-Time Video Understanding via Cross-Attention","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-28T15:22:31.310003Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2606.07639"},"observation_digest":"sha256:85e96780b04f57d7ee1312527bcad38f27a94d0bac2bbdc134d636dedeec4523","observation_id":"b4edbaed-a809-41bb-8800-3283032b198b","resolution":{"observed_at":"2026-07-01T22:26:17.943893Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2606.08615","last_updated":"2026-06-07T13:00:19Z","snapshot_observed_at":"2026-08-05T22:03:12.300616Z","submitted_at":"2026-06-07T13:00:19Z","title":"Harnessing Streaming Video in the Wild","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-06-27T18:47:55.910417Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2606.08615"},"observation_digest":"sha256:8f0544f3b2befb0a726c789157a7c5e2f2002f6ec57cb36bb0322883e8026367","observation_id":"e5ffa6bf-aca8-44cd-a9d1-edb487d90560","resolution":{"observed_at":"2026-07-02T22:37:25.724490Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2606.12195","last_updated":"2026-06-10T15:17:08Z","snapshot_observed_at":"2026-08-01T02:09:41.655807Z","submitted_at":"2026-06-10T15:17:08Z","title":"InternVideo3: Agentify Foundation Models with Multimodal Contextual Reasoning","version":1},"reference_index":239,"source":"arxiv_source","source_observed_at":"2026-06-27T09:48:27.652901Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2606.12195"},"observation_digest":"sha256:3739c344a092f7bf1472c006304ed3ba321f99272367277546b9974d4a891858","observation_id":"d8935e4a-e06f-4361-9c56-c5bb483fa60d","resolution":{"observed_at":"2026-07-03T10:48:02.954765Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2606.21734","last_updated":"2026-06-19T20:43:49Z","snapshot_observed_at":"2026-08-05T18:05:51.515234Z","submitted_at":"2026-06-19T20:43:49Z","title":"HPP: Hierarchical Programmatic Probing for Long Video Understanding by Decoupling Perception and Reasoning","version":1},"reference_index":214,"source":"arxiv_source","source_observed_at":"2026-06-26T14:19:53.450263Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2606.21734"},"observation_digest":"sha256:8e5c69d925a28c0b7e6715ec6aa95c7c02a77e6d3b0248c8c02b9f97d7fbe104","observation_id":"9b64af0e-6ea9-4c66-b1fb-991a5e64e76a","resolution":{"observed_at":"2026-07-04T06:39:37.594278Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2606.24253","last_updated":"2026-06-26T09:37:54Z","snapshot_observed_at":"2026-08-11T09:58:16.500600Z","submitted_at":"2026-06-23T07:42:22Z","title":"TuringViT: Making SOTA Vision Transformers Accessible to All","version":2},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-06-29T05:32:26.746776Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2606.24253"},"observation_digest":"sha256:15bb0892fb6792a0c5b2ba7a39f9a5a4bd9e451373eaa81d0645e88393ac11da","observation_id":"94c60c0d-7c65-4543-9738-6d11ffa630f3","resolution":{"observed_at":"2026-06-29T15:03:32.153015Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2606.27999","last_updated":"2026-06-29T10:15:53Z","snapshot_observed_at":"2026-08-06T16:35:46.841389Z","submitted_at":"2026-06-26T11:52:37Z","title":"HumanMoveVQA: Can Video MLLMs reason about human movement in videos?","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-29T04:53:11.830488Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2606.27999"},"observation_digest":"sha256:f4e55196bdfcca2257363b0ba5bb3e55a0a8d49330b3496da3626e2bd01a46a6","observation_id":"ea97fda3-dbeb-4a0b-b958-904cbba4fd93","resolution":{"observed_at":"2026-06-29T19:13:53.053611Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2606.27999","last_updated":"2026-06-29T10:15:53Z","snapshot_observed_at":"2026-08-06T16:35:46.841389Z","submitted_at":"2026-06-26T11:52:37Z","title":"HumanMoveVQA: Can Video MLLMs reason about human movement in videos?","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-30T09:46:56.063333Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2606.27999"},"observation_digest":"sha256:f7ff27dac4bd4b319ff9eea18cadc2b456cf78f6cf5ed11eed05f3a3a2e947d9","observation_id":"e8d11c16-d725-44f1-83f8-faa35cb9e3d2","resolution":{"observed_at":"2026-06-30T09:54:35.321646Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2606.30220","last_updated":"2026-06-29T12:33:15Z","snapshot_observed_at":"2026-08-07T20:02:15.815492Z","submitted_at":"2026-06-29T12:33:15Z","title":"From Accuracy to Visual Dependence: Auditing and Filtering Modality Collapse in Traffic VideoQA","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-30T06:14:11.109840Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2606.30220"},"observation_digest":"sha256:7e3e1f3a5bbe3784030f40955fd0f437230dee104c9da6f5dc0e2369bf0ee368","observation_id":"e61ca1cd-6bcb-4d80-80f7-7efbe79f15d5","resolution":{"observed_at":"2026-06-30T06:14:18.602974Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2607.00867","last_updated":"2026-07-15T08:03:34Z","snapshot_observed_at":"2026-08-02T09:17:20.227144Z","submitted_at":"2026-07-01T12:32:29Z","title":"EFlow: Learning Evidence Flow for Long-Video Reasoning with Adaptive Reflection","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-07-02T14:19:04.890074Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2607.00867"},"observation_digest":"sha256:30d00bde753fed90410180fe89397faeb0aea5b51b38fcf701089f548b96a6ca","observation_id":"764730a7-0eef-4edd-ad3c-c58cb9d0d784","resolution":{"observed_at":"2026-07-02T14:27:03.283511Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-07-12T09:13:38.446621Z","title":"Jihan Yang, Shusheng Yang, Anjali W","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.00867","last_updated":"2026-07-15T08:03:34Z","snapshot_observed_at":"2026-08-02T09:17:20.227144Z","submitted_at":"2026-07-01T12:32:29Z","title":"EFlow: Learning Evidence Flow for Long-Video Reasoning with Adaptive Reflection","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-07-12T09:13:38.446621Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2607.00867"},"observation_digest":"sha256:bafe48ed765b8739bcb56741537177fc115e4512daae9103e39eccf4edf85956","observation_id":"c35bc9f5-5714-468c-a5f2-465057f61b41","resolution":{"observed_at":"2026-07-12T09:13:38.446621Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-02T09:17:20.580137Z","title":"Jihan Yang, Shusheng Yang, Anjali W","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.00867","last_updated":"2026-07-15T08:03:34Z","snapshot_observed_at":"2026-08-02T09:17:20.227144Z","submitted_at":"2026-07-01T12:32:29Z","title":"EFlow: Learning Evidence Flow for Long-Video Reasoning with Adaptive Reflection","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-02T09:17:20.580137Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2607.00867"},"observation_digest":"sha256:6932f4eae2d3fc5d1707d1fc05bf46cbb40e92d67bbd02da8e538e0262f019e3","observation_id":"58b91602-6fc8-424f-a71e-b6fe58998e8a","resolution":{"observed_at":"2026-08-02T09:17:20.580137Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":"2407.15754","doi":"10.48550/arxiv.2407.15754","metadata_source":"pith","pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","venue":"cs.CV","work_id":"f6bb6dd9-35f5-4788-9063-af4d76699147","year":2024},"citing_paper":{"arxiv_id":"2607.01751","last_updated":"2026-07-02T06:07:44Z","snapshot_observed_at":"2026-08-04T13:31:05.282842Z","submitted_at":"2026-07-02T06:07:44Z","title":"MedStreamBench: A Time-Aware Benchmark for Streaming and Proactive Medical Video Understanding","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-07-03T16:37:09.666491Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2607.01751"},"observation_digest":"sha256:6a8141181672440929c2b2ae95dc06ca1bbeb170f700e16b279b779bfaae8364","observation_id":"440bf13e-0fa7-40d0-b7b4-d37f5da42a69","resolution":{"observed_at":"2026-07-03T16:38:39.533601Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-07-12T11:31:14.532101Z","title":"Longvideobench: A benchmark for long- context interleaved video-language understanding, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.02551","last_updated":"2026-06-26T16:05:57Z","snapshot_observed_at":"2026-08-06T08:59:01.715141Z","submitted_at":"2026-06-26T16:05:57Z","title":"DELTAVID: Enhancing Fine-Grained Spatiotemporal Perception with Cross-Video Differences","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-07-12T11:31:14.532101Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2607.02551"},"observation_digest":"sha256:8e472ff96bd7b297be02317984889d0243a2f8e3f5efd1f20dcaea42e995c1ac","observation_id":"a41d2246-b9ea-4d8b-a2e1-83d53d033fdb","resolution":{"observed_at":"2026-07-12T11:31:14.532101Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-02T00:44:49.338761Z","title":"Longvideobench: A benchmark for long-context interleaved video-language understanding, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.14935","last_updated":"2026-07-16T12:47:59Z","snapshot_observed_at":"2026-08-09T10:08:14.840784Z","submitted_at":"2026-07-16T12:47:59Z","title":"VideoChat3: Fully Open Video MLLM for Efficient and Generalist Video Understanding","version":1},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-02T00:44:49.338761Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2607.14935"},"observation_digest":"sha256:c11c52eac4f6735e2d4c2e2f347d630602b17bd8a9afd4951570e9a36c3e10fb","observation_id":"43fc2d0a-4e5a-49ff-8900-0cd58958772e","resolution":{"observed_at":"2026-08-02T00:44:49.338761Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-07-31T06:20:14.090705Z","title":"LongVideoBench: A benchmark for long-context interleaved video-language understanding.arXiv preprint arXiv:2407.15754, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24904","last_updated":"2026-07-27T17:59:53Z","snapshot_observed_at":"2026-08-10T20:15:55.135228Z","submitted_at":"2026-07-27T17:59:53Z","title":"Mage-VL: An Efficient Codec-Native Streaming Multimodal Foundation Model","version":1},"reference_index":120,"source":"pdf_text","source_observed_at":"2026-07-31T06:20:14.090705Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2607.24904"},"observation_digest":"sha256:052d4ddefdce8c04d00956848f4afb78bec64de2de45eeb716e021437053ff0b","observation_id":"f0c34a32-a501-489c-b451-3a44d115b350","resolution":{"observed_at":"2026-07-31T06:20:14.090705Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-04T23:46:51.599549Z","title":"Advances in Neural Information Processing Systems (NeurIPS), Datasets and Benchmarks Track , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.01644","last_updated":"2026-08-03T03:27:46Z","snapshot_observed_at":"2026-08-10T14:57:44.819541Z","submitted_at":"2026-08-03T03:27:46Z","title":"CRAFT: Compression via Recursive Adaptive Fusion of Video Tokens for Vision-Language Models","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-04T23:46:51.599549Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2608.01644"},"observation_digest":"sha256:9c60ba749ec962557b6c66dd5b59c930a8b3feb030dc14a7cddc67e92040a74c","observation_id":"1c803385-804f-4d59-9e6e-d27819625dd2","resolution":{"observed_at":"2026-08-04T23:46:51.599549Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-04T17:24:14.910935Z","title":"2407.15754 , archivePrefix =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.01980","last_updated":"2026-08-03T09:42:42Z","snapshot_observed_at":"2026-08-11T15:40:04.127681Z","submitted_at":"2026-08-03T09:42:42Z","title":"AdaThinkV: Adaptive Thinking for Token-Efficient Video Reasoning","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-04T17:24:14.910935Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2608.01980"},"observation_digest":"sha256:d01a1b4c4499a1a3c942bdc6f76d859a1fb2c85770254bf2da88b9326fcd138d","observation_id":"6d0275f8-b6b2-4745-a2f3-52fe59b7d08a","resolution":{"observed_at":"2026-08-04T17:24:14.910935Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-08T20:16:34.505752Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.04302","last_updated":"2026-08-05T00:20:23Z","snapshot_observed_at":"2026-08-10T18:05:29.552100Z","submitted_at":"2026-08-05T00:20:23Z","title":"CLIP-CC-Bench: Evaluating Paragraph-Level Video Descriptions in Video-Language Models","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-08T20:16:34.505752Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2608.04302"},"observation_digest":"sha256:bc7c26e1b7700e118420f711165fdfbd27cba1a914f6745f39e856160e85c622","observation_id":"d072fc63-4471-401d-bf6e-e8000cacb45b","resolution":{"observed_at":"2026-08-08T20:16:34.505752Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-07T04:24:54.793516Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.06361","last_updated":"2026-08-06T17:57:06Z","snapshot_observed_at":"2026-08-10T12:08:17.898559Z","submitted_at":"2026-08-06T17:57:06Z","title":"The Low Frequency Trap: Video Language Models Fail at Simple Event Bookkeeping","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-07T04:24:54.793516Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2608.06361"},"observation_digest":"sha256:c78343231d8dd98247bad4d5ec7487c759c07e33df8886eefb5c8630442a92b0","observation_id":"abc2a3de-42ec-41a4-9a5c-bffa94bcc04d","resolution":{"observed_at":"2026-08-07T04:24:54.793516Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-10T21:18:34.640153Z","title":"Longvideobench: A benchmark for long-context interleaved video-language understanding.arXiv preprint arXiv:2407.15754, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.06756","last_updated":"2026-08-07T03:24:25Z","snapshot_observed_at":"2026-08-11T18:10:44.168564Z","submitted_at":"2026-08-07T03:24:25Z","title":"Capek 0.5: An Execution-Centric Vision-Language Model for Embodied Intelligence","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-10T21:18:34.640153Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2608.06756"},"observation_digest":"sha256:9a3879ea7008c29183148b2299fa1d55ea2ffb97bf2002128f7de59769d49846","observation_id":"d88f57ac-3e11-4f7a-b937-26d411effd17","resolution":{"observed_at":"2026-08-10T21:18:34.640153Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2407.15754/citation-record","integrity":"/paper/2407.15754/integrity","json":"/paper/2407.15754/citation-record.json","paper":"/paper/2407.15754"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2305.06500","last_updated":"2023-06-15T08:00:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-11T00:38:10Z","title":"InstructBLIP: Towards General-purpose Vision-Language Models with Instruction Tuning","version":2},"cited_work":{"arxiv_id":"2305.06500","doi":"10.48550/arxiv.2305.06500","metadata_source":"pith","pith_arxiv_id":"2305.06500","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InstructBLIP: Towards General-purpose Vision-Language Models with Instruction Tuning","venue":"cs.CV","work_id":"f3aac728-ded0-4e55-aa9e-4a1635d4313d","year":2023},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"cited_paper":"/paper/2305.06500","citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:7a8c8a54a18e934bf6ab5d73ee693d7759bb4dfcc7d7aa71c28e7ea419f90583","observation_id":"9778ac48-8944-47cb-8e6f-258cc0d91a87","resolution":{"observed_at":"2026-05-18T00:30:12.236112Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-07-12T21:49:51.838916+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T21:49:51.838916+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"e6317a51-8c7e-4ff7-be7b-38e9e251b03e","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:7b232cd334976f1847d19f882dfbe01b7bb61ed9437f33b3ad3d602740febd8e","observation_id":"aa581938-1694-49b2-8a0d-d5c81a5d3aa5","resolution":{"observed_at":"2026-05-18T00:30:12.240072Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"(a) Did you state the full set of assumptions of all theoretical results? [N/A] (b) Did you include complete proofs of all theoretical results? [N/A]","venue":null,"work_id":"ae0447d1-dd72-4adf-a573-6ec340117941","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:7f28ab8666dc38385b605fdd1a1f43aa4f927e58c39f66736c60698fafbdf31d","observation_id":"5e3c471b-e652-4fa8-b987-b3cf93fd9edf","resolution":{"observed_at":"2026-05-18T00:30:12.244751Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"for benchmarks)","venue":null,"work_id":"7f7388a3-739e-4364-951d-da877ab4ee6d","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:ba5b24b2661e7b442dec50635c4d4c8d806c93e32e0a7f55ab128e82a657d01c","observation_id":"4686ddf8-6d03-405f-b2a8-fd4068496ca8","resolution":{"observed_at":"2026-05-18T00:30:12.248747Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"4060be48-948d-492f-aafe-d9f8da11297e","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:11e31db9c4b20c88c24de91cc89912d7a4386eb166fe45a719dc13c9fb47ea22","observation_id":"208e8b85-27dd-47b2-93b3-fc7405a36794","resolution":{"observed_at":"2026-05-18T00:30:12.252489Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"type\": \"image_url","venue":null,"work_id":"c6c74d19-8773-43f7-9a10-41ac915ce539","year":2024},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:5bb80a6d448e15dfa64d54229a2214580b29256c966a0206da401ce6815ed095","observation_id":"e17ff800-ba92-445a-9d91-4bc0dbbb7210","resolution":{"observed_at":"2026-05-18T00:30:12.255759Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"ce81c77c-b11d-45ec-9816-fbdb66a5ec45","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:a662d138153bc743f3eb5accdd2cd3f57feaf5eab6363c12d2a1f7cb5794672e","observation_id":"a76c3055-df89-49a2-8d17-af72f6ae45ac","resolution":{"observed_at":"2026-05-18T00:30:12.259418Z","resolver_source":"raw_fallback","status":"parse_uncertain"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"66e88aa8-37ac-4b7e-906f-5e257e9d5b2b","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:d8612314b08b9fb67a171852d83b1577eb251aa993b3116aecdfb004cf6db072","observation_id":"0e4c9b66-aacc-4f68-9d63-722718fd4b22","resolution":{"observed_at":"2026-05-18T00:30:12.262714Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"548e791e-1627-44ef-956b-bbd476e39d70","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:6e8c61ebb58de3f0eebad8d0169afec0327d6fad6b0a1f932c62f98abf064062","observation_id":"e357e29a-c130-453e-8795-18fd806ac2dd","resolution":{"observed_at":"2026-05-18T00:30:12.265672Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"S CENE -REFERRED OBJECT (S2O)","venue":null,"work_id":"5d411f4a-9160-4f08-bf97-a5ba40f722bc","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:5d026f4ce87c9b68197da71d8e15ce479397f84edcee2bd59cf8c2404389f8fa","observation_id":"9d5561d4-55df-41c3-95bb-47ad83b04603","resolution":{"observed_at":"2026-05-18T00:30:12.268525Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"8b1e6158-6790-4ec4-8df4-5150ed7e81ec","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:06ce5c6eba3acfc1f79673b162a5e7195b3271f1ac3a75a57aae43a5f490c429","observation_id":"46016355-a204-4ebd-950f-3b70e34c7acc","resolution":{"observed_at":"2026-05-18T00:30:12.271080Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"606c03f4-33bd-4301-aacb-320bc4371774","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:7157b95023bd13525a7af559bb6bb7c78f95b2d5e4517f7e364802fe0e781fb2","observation_id":"4a2dcf99-66cd-4768-a06f-908a2899c9ac","resolution":{"observed_at":"2026-05-18T00:30:12.273845Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"S CENE -REFERRED OBJECT ATTRIBUTE (S2A)","venue":null,"work_id":"14ff53f0-0e69-4d51-a196-68a2f7740913","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:db9ccf708b84a498696d16fc03b2bed6199dec40b0d13254a789e517c76b7332","observation_id":"60dcdbe6-2cae-42a9-b8d6-2d9f28721389","resolution":{"observed_at":"2026-05-18T00:30:12.276814Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"8a6d9960-34a8-4fe5-b9c8-86cc5be4a881","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:5def0f812f1eef964eec6fa7491cf869f0491497a522e0a21437d66ba10c129d","observation_id":"50ffa68f-db10-48e7-9a43-8ad2c213cfe1","resolution":{"observed_at":"2026-05-18T00:30:12.280082Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"114e19c9-a2d6-48e4-b5a0-0b1edd353176","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:9200bf428164209d2fce55b82b5fe3cb46186f079e6aa28e60b45ed85f8468bf","observation_id":"f0cc634c-8c3a-41b8-8f55-efd8c80fa748","resolution":{"observed_at":"2026-05-18T00:30:12.283441Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"1bf8897a-686c-48d6-9477-aee89b57ecd5","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:26303801152a7c609cd5280e4a89c7689c16199496d4dc2963a35bbea63760a6","observation_id":"bf92060a-f2de-4c54-82c2-fb2508ae21b2","resolution":{"observed_at":"2026-05-18T00:30:12.286936Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"EVENT -REFERRED OBJECT (E2O)","venue":null,"work_id":"8019504b-a94a-4029-9308-b754470a64dc","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:0740cc17d908d8a4ab1a13e428fc1189a70c088ad24f84769192faa562abfd2b","observation_id":"72014946-0eef-49f0-b4f1-ffa1479eead4","resolution":{"observed_at":"2026-05-18T00:30:12.289807Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"47c9303f-ff64-4737-a386-589a65e162d9","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:f32ff7f155dbdd81e879b58f015dbd38f88b6cbd75fec2fe85cc393f782abcd5","observation_id":"9588a877-7426-4fc8-9943-26c6896fe20b","resolution":{"observed_at":"2026-05-18T00:30:12.292436Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"175ef494-750a-40be-b35b-3d62616ddff9","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:4d6e1f0f3c3d2ddb175b47b92f13ee3c1a0a1419fb499d7200c1780f6ebe5099","observation_id":"e95480e1-8263-48e7-aa63-3e5f884117eb","resolution":{"observed_at":"2026-05-18T00:30:12.294860Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"ff8e3a0f-9669-4181-bd55-fa988861ca01","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:a9599734b82c3ed7dd5f16d8dd40f3657f9cd61648b193f103771831f1ecb1ea","observation_id":"1b47bc9c-09d2-493f-a5ad-577e6c24c99a","resolution":{"observed_at":"2026-05-18T00:30:12.297613Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The options should also be as detailed as possible","venue":null,"work_id":"3274d5c8-4db8-40ca-944b-9c2e01a3a4c8","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:f6dde66fd2515575e2452d78206534cbe400e876f85b0d6583e42f9c58ed9097","observation_id":"b252bc85-7042-487f-8a9c-ae19f37622d5","resolution":{"observed_at":"2026-05-18T00:30:12.300253Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"OBJECT -REFERRED EVENT (O2E)","venue":null,"work_id":"a1d20475-09fa-489d-8187-95acc644d5fb","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:2e6d828ee0baca9a37f42e16fc9e41de0df51f7902ed22914dce475c1914a04f","observation_id":"2502aa5d-639f-4ded-8b07-50753764f847","resolution":{"observed_at":"2026-05-18T00:30:12.303040Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"fa556d44-2a4e-4c20-aa87-0fabd7125fb3","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:f3bd1dda0feaa3c6f58513e03fad8cde4f248f34a8140814ab43b37c4fdeea94","observation_id":"451a23d5-87c9-48cd-a5ee-62a1f61e1480","resolution":{"observed_at":"2026-05-18T00:30:12.306070Z","resolver_source":"raw_fallback","status":"parse_uncertain"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"dc9daed8-2ee6-46a8-b1af-ea3d6930d7f5","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:b3372ba4172cfe2b3117f3698d65528c238211021e1bac9d7b25ee9f67fcac6c","observation_id":"f027944b-4e1c-4610-bcf5-8ba5e754e96e","resolution":{"observed_at":"2026-05-18T00:30:12.308833Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"64d6a5d8-84e3-4fbd-86bf-857a9006e760","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:0689569b67687e07521edc93440aeb0116c60797fa2ae35a5d2c3320712756de","observation_id":"59db06d9-a26a-4638-a896-5660c4729578","resolution":{"observed_at":"2026-05-18T00:30:12.311709Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"079ce4a3-952b-4116-9a4c-4624473ac90a","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:aeb9a48deba2f7372a40c18d3b15733e99c196ffdba137149a690748b6ec2e60","observation_id":"460b0e1b-4560-4a2a-a467-472fe30610be","resolution":{"observed_at":"2026-05-18T00:30:12.314464Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"T EXT-REFERRED EVENT (T2E)","venue":null,"work_id":"3a0ce875-2a43-4be0-b795-2a5fcc8ada98","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:d2d4c5b0d5725c597f8a9273cf7c3ee3e45187e265407edb9dc729644d867a8e","observation_id":"d8f715f0-742b-4020-bf43-a2c695273655","resolution":{"observed_at":"2026-05-18T00:30:12.317600Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"ecadeba1-6a0d-481b-beb2-121f39b17120","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:c3add600cb4e6e86bfcbd025e482bd166341bc2ef992c719291fc5945130273b","observation_id":"86daad49-8773-4206-aacf-db88a91b5523","resolution":{"observed_at":"2026-05-18T00:30:12.320376Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"72915b83-a69f-41e2-8840-fc31eafbf83e","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:4565745ca92cc9577fbad738b4a9f339c09a7a1224a7fe31536898f479c8f39e","observation_id":"c884ea5d-6812-4fd1-acc0-bd44d20eadf4","resolution":{"observed_at":"2026-05-18T00:30:12.323249Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"d1865445-c57b-49d1-917d-cbaf58ca57cf","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:2801f229531f0aca706043de1bb2da64606b97b61e64d8a693b0012f6689c49b","observation_id":"ea3dd964-7f10-4e6f-a0cc-96e749de6332","resolution":{"observed_at":"2026-05-18T00:30:12.325993Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"bidirectional encoder","venue":null,"work_id":"74be7c01-1e33-4d73-93a9-64a56b087692","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:3e4cf0d083dd5774566876bff365a891074e6b5baaa4e725ea48266670f6ffb4","observation_id":"3d0a6199-c30a-486b-9003-54fc9e02b1b0","resolution":{"observed_at":"2026-05-18T00:30:12.329186Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"14510773-b84b-48e9-ab4b-224ac9e5330e","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:ca189363872b451aff8805e660ebe1149f17fc5a6e21bd3ca6c0cb84e99c878a","observation_id":"fa7cefe8-4f9d-4be0-8a46-498d98d34107","resolution":{"observed_at":"2026-05-18T00:30:12.332524Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"0e17aea4-8c70-4860-8504-51e322d3ca96","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:534511d36eb6e8ddfac752b5fd2dc797670e120d890a39a4015a824f4ad89e62","observation_id":"674e5fb4-b2d7-4930-9b80-4ab86f9123d6","resolution":{"observed_at":"2026-05-18T00:30:12.335197Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"5ccdddf6-80cd-400c-ba73-6b6f25fc49ce","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:2294f94216ef3003b855743daed4f139af7662cbc8c3947b10a76ebd9bbb3849","observation_id":"e7b62866-2161-4ffc-a99b-7dfb25d15b9e","resolution":{"observed_at":"2026-05-18T00:30:12.338261Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"revolutionary changes","venue":null,"work_id":"c179d357-0316-413d-8676-a7bbd94c3d34","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:e4a19f54658ca7fce615ce1e82b8366fd14b27b8cc9b5014152f3dc956396a58","observation_id":"0deebc30-e593-404a-8365-0174a555470f","resolution":{"observed_at":"2026-05-18T00:30:12.341045Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"4a24442d-8049-408d-aaf8-216d13d17b9a","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:b92423f6dc04d49906040bac4f543583c3a89d46e05fffd9b8b9d5d2150b156c","observation_id":"175394b3-2f44-4a36-925a-551c959ef776","resolution":{"observed_at":"2026-05-18T00:30:12.343628Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"85ed0ff1-f642-4dcd-b5b3-81629481bf66","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:768e6d66778060f18632360c9579bde365cf1dff23e8b8fe7889363ef819b3b0","observation_id":"5b15de22-e16e-4766-970c-300d5ad6abfe","resolution":{"observed_at":"2026-05-18T00:30:12.347509Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"5e042152-a997-4ae4-90d5-65f7339a8ad8","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:3b1da546d2fa43c67299e380f5dc29c6c016d5ccd2fd4df4dde5cc6bfdf6303d","observation_id":"546940db-fcf0-4f20-b887-65effbfd8a33","resolution":{"observed_at":"2026-05-18T00:30:12.350214Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"b76eae1a-1645-465a-93b7-5dbd1309d986","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:9db169f39f99fbdc060964aad40323bbdc5da1517de4e384721c48a066c1ab99","observation_id":"57851ca6-89ac-4d68-a3ec-16ba710b2bd2","resolution":{"observed_at":"2026-05-18T00:30:12.353134Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The specific instructions for each category of (L2) questions are as follows","venue":null,"work_id":"852d3f03-8ac1-4da2-aee0-32cfdcf5abf3","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:b2effe481dad2385fc55f7204d3d1a9110c1494d5a22dca423472b41dc3b23d4","observation_id":"077ae30f-319e-444b-9758-8a1b14bebae2","resolution":{"observed_at":"2026-05-18T00:30:12.356872Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"c45b0d10-c4ac-4db4-8916-a1dcd0f2feba","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:fd854c3bcc2f1fc688aaccea1ec9a6df3c6cd10944aa2ca158e92b55e8cc9f35","observation_id":"a7ee1bd7-7f26-43d3-b626-edc0cd042712","resolution":{"observed_at":"2026-05-18T00:30:12.360440Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"55eba80d-d665-44fe-a60d-e54dd1750e8e","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:374530b9233a6f342a9223b786403d4881f3cc25859ba1185bb0805e689d3725","observation_id":"8682e29f-27bb-49d6-97ad-d3a4f87a3f9b","resolution":{"observed_at":"2026-05-18T00:30:12.364838Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"O BJECT BEFORE /AFTER OBJECT (O3O)","venue":null,"work_id":"b3507218-6def-4edb-91c5-e1bad171675f","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:f75d5c0cd0657b34202396f85765da50cc5c3f603a1c96dbee5598fe335862aa","observation_id":"332c80fe-b556-433a-889e-6662556de7e1","resolution":{"observed_at":"2026-05-18T00:30:12.370133Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"8a111c1c-136f-4988-9278-0969ef6dc0d7","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:abf3549d5a407a4a856ca980a205df1003f91b2cbb29ec88133158be81df7702","observation_id":"eaa2d8ce-76fb-4868-a944-0938cca749a9","resolution":{"observed_at":"2026-05-18T00:30:12.375014Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"85617f0c-c2c7-42c1-a969-de4dc14b27d1","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:5c1b166d27543b2a810f5ffcb3cbf4701ab09228a4618994b38a558b2d137050","observation_id":"cb6077ea-8dba-4efd-92e2-39017295c626","resolution":{"observed_at":"2026-05-18T00:30:12.379429Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"S EQUENCE OF SCENES (SSS)","venue":null,"work_id":"2c22df59-792f-49eb-be6d-a55613f6c02d","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:923cf7fddab70c4317fe402237cfc12c9ddbe2d2351a0052ad903c4113121620","observation_id":"8bf19a34-6018-4426-a4aa-c561442c39dd","resolution":{"observed_at":"2026-05-18T00:30:12.420491Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"507da9e1-6eb3-4b5a-9282-0146405a5f10","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:87da87f879d00137f148b385fdbf2db70376951986311743a71904eab6e9740e","observation_id":"2c18883f-ac04-4e8d-a205-0bde7dbd1a85","resolution":{"observed_at":"2026-05-18T00:30:12.424207Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"c5dcf379-2e66-4332-8964-485a8b5914ed","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:a9dddea54cb10032dbe18dfa479f617496dd5eca0f22343fa0f8651031c72758","observation_id":"edb2339e-e28e-4400-ac1b-f38a4ab7971d","resolution":{"observed_at":"2026-05-18T00:30:12.427520Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"d3db049d-2cdf-40d1-b728-2801699845f1","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:ad45d23937bd425b9b1f187ebcd4a236888a2226e75337485d6a54acb27b5dd0","observation_id":"e3b2fa75-e789-4589-9236-33fcdef64404","resolution":{"observed_at":"2026-05-18T00:30:12.430610Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"First, a segment of the experiment video is played, then slides with text are shown, and finally XXXX","venue":null,"work_id":"cc30c2b1-9fd9-4ed1-8971-d87339c30477","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:5e56d0539e6b8179c9d0c0c3f006fa8bce3be60c6844a44940777104face671b","observation_id":"3c09f3f5-b940-4714-8626-948a5a144a45","resolution":{"observed_at":"2026-05-18T00:30:12.433742Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"85cee253-c7d2-4346-996c-05e8d88ab6cc","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:ff351cedf99f47466338e57932023bcd3c146873e3f591534cea4ac44394dfba","observation_id":"da006f10-b2bd-4144-b26f-02d3a93c8a1e","resolution":{"observed_at":"2026-05-18T00:30:12.436475Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"67dbec85-042c-4b1e-84de-ad8886e20f25","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:ea1e77ab77a06088fc28b93bc41386bc260b2e88bc6f23ca260a5990626776fd","observation_id":"09c1e360-1473-4b6e-a08b-ee9297a37b14","resolution":{"observed_at":"2026-05-18T00:30:12.441203Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Square on a sunny day, – B","venue":null,"work_id":"a107fa23-7442-4e4e-883a-ce63ebd8815d","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:a90ab37cf55718f70327b994432b820cec21c02ffb069f675a1a1a566504cbf1","observation_id":"cf154d86-38a8-4ebf-a661-cfe0b8b84014","resolution":{"observed_at":"2026-05-18T00:30:12.444305Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"3178c005-f67f-487e-96f6-d09fda8e560c","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:c668c0d7a72ebbf299d5f116f098e9425a38822a8b8231b0efe524357cafd9b0","observation_id":"5cb79dad-52de-4f9d-9a3c-1f2bb73aec4f","resolution":{"observed_at":"2026-05-18T00:30:12.447085Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"5c182647-1c14-4363-8b40-834691f282da","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:bf4362e67b3607983370f3409de16dddd90fce32ee18692f6f638b1236ba5239","observation_id":"a443d663-3013-41ae-8292-f6ebd0911e62","resolution":{"observed_at":"2026-05-18T00:30:12.449781Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Changed from a white T-shirt to a black vest – B","venue":null,"work_id":"1f06a28e-767d-4186-b32d-91946356d33a","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:3ad0bc0c95d4740266c264e8292c39ecb5b0e38bdf39aab9509ea64f7dcb56aa","observation_id":"bc575d9c-a1ef-4646-b270-a249942b8746","resolution":{"observed_at":"2026-05-18T00:30:12.453087Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"b2735cde-be7f-44e5-aa8c-5101f541d981","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:a2cab111833aec5a0221f4550d96322ebd3eb781cebb7e86d2b51caf020780ec","observation_id":"a6cc860e-fd61-4cd9-8b82-d2a745f1a600","resolution":{"observed_at":"2026-05-18T00:30:12.455895Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"7b13b65d-6087-4fcb-9716-bbd826315202","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:95e9ead4101e8f71c7f386485486a92d58f5c89c14d5dcf15c5f34f18c5a09c8","observation_id":"10f79ac5-a18a-4f49-9187-2ae2a114eec1","resolution":{"observed_at":"2026-05-18T00:30:12.458742Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"30f2ff8e-848f-43c2-99f7-5ce60831d2d7","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:fc0b35aec50568d2b9395e5d24bdd2b0e8407eaf098c71f7b0cfd6ee8af5c75d","observation_id":"f6c2e0d3-19fc-4660-a0d7-2ed8f930a047","resolution":{"observed_at":"2026-05-18T00:30:12.461666Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"I eat an apple every day","venue":null,"work_id":"246b834e-d353-455d-96c6-0c99d1d34938","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:2bd7b91a427b3ca2f129f75ebead1e61bb29908757590ff425bc8e6adc54fb19","observation_id":"081bced1-1f62-4de0-8695-e217fb9014d8","resolution":{"observed_at":"2026-05-18T00:30:12.464448Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"709862a1-43ff-40cc-8eff-0bc4e478ed8c","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:c49fd05c4aae5e9db3a3b887a878bb0b170faa3c1b0af3ee8e16783af58f154b","observation_id":"c683be6a-a681-4853-8996-0d64f068e32a","resolution":{"observed_at":"2026-05-18T00:30:12.467362Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"a738ee48-4400-4216-926b-77616a56b291","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:03a26247a2d1e3c2781d4cdca8f3dd2f93236e339878d9514487a77a3d2746b6","observation_id":"3a4ae5b3-a99c-4d33-b589-1acdbf4fc8b5","resolution":{"observed_at":"2026-05-18T00:30:12.470817Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"b5e5e5da-64f0-4eae-a70c-fa6fbed9fffc","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:d0ffce3ca1d620392fa93427aed57e23e51c853a53f7671726b72bb5429d646f","observation_id":"99341c00-171d-4e09-868e-965a550151a3","resolution":{"observed_at":"2026-05-18T00:30:12.473478Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"100 years later","venue":null,"work_id":"9ddaea75-b6c4-4e78-a447-f6bf3e882b4b","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:814b28ccf9fb5d0d5fbe796ce2c42c67607e351cd7c88d06123e21ccfe663c57","observation_id":"1c39d814-416c-44e1-a61a-f21932e6a694","resolution":{"observed_at":"2026-05-18T00:30:12.476443Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"dee7cea7-a38b-480a-b94a-df8a431ed2b4","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:29de664dd98f4d858493b3d1e9bd39eef8cf24e63976376f1c138a77ffda2529","observation_id":"f9aedfcb-851a-492d-a7a5-a4f5ed49ccca","resolution":{"observed_at":"2026-05-18T00:30:12.479602Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"cb12ea12-84c0-4fae-b3c7-3dda5c1456f5","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:df42d2ab0c6d790600d147cf2dfa09909f522ea20657ff57534aeea4f78abca9","observation_id":"89e08865-b714-4465-8a26-0b0acca26544","resolution":{"observed_at":"2026-05-18T00:30:12.482237Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"676ff63b-5b05-445e-ba69-72b18b5c1fd3","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:bd7baba175dbd80e25b225a67fde51c797faba6cf9dce4fae6d88333d5d21bf4","observation_id":"3caf0330-8dfc-44b5-b093-5c2e1a3900c5","resolution":{"observed_at":"2026-05-18T00:30:12.487308Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"T EXT-REFERRED OBJECT ATTRIBUTE CHANGE (TAA)","venue":null,"work_id":"481fe813-cf03-4506-b97c-31288267ad5f","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:935cd756491ae4243814e9d45bbea822433355b089663ac471d12bc391f6716d","observation_id":"ad199cda-3775-4501-adee-1baf98289596","resolution":{"observed_at":"2026-05-18T00:30:12.490356Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"21 Figure 6: The annotation interface for L ONG VIDEO BENCH","venue":null,"work_id":"f019f2c2-b004-4d8d-9a30-6a85dd2cc646","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:7837bfa603646fc306ab03de2d645b20aad9a8885a218a2ae178013cdaf9d0a2","observation_id":"e81267e2-af6d-4fab-9fb8-24dcb52290e2","resolution":{"observed_at":"2026-05-18T00:30:12.494097Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"249447f9-44af-4c83-bb63-9ec0aa0182bb","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:0e0fb531dafa82afa56698d91d19328a8a09fa44a6fa34c9dac568219c9a624d","observation_id":"dd7db6ab-de13-47e4-8f66-19ea9d9fd8f7","resolution":{"observed_at":"2026-05-18T00:30:12.496930Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"600c189d-3a3f-4440-ae95-2c58778945ac","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:fb17b2f19bed781b6bfaf2d53a9becbea72abfdb633fa573e2cb9c865a59e5ba","observation_id":"9c6cb916-7ef4-40f2-9199-566b1af1e3fd","resolution":{"observed_at":"2026-05-18T00:30:12.502926Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"I am going to sleep","venue":null,"work_id":"ab494856-59e3-4b1f-994b-1e0ae87e7de8","year":2024},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:8e9349abd180442c4abf196cf982ed4bbf76e7b0131e79479c6f0daf56b730f2","observation_id":"cdf15c9e-b554-42be-9246-0e992d33b1a2","resolution":{"observed_at":"2026-05-18T00:30:12.505993Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"aea4a2ee-0863-47cc-b7b3-d74067e00193","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:b33f8ef8a89e8c31b22d8bf05abdf22ae36d10c3df0fdae6ba7e362fa2afc9dc","observation_id":"5f5fab09-050f-4880-ac5b-b8ce4cffe41a","resolution":{"observed_at":"2026-05-18T00:30:12.508824Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Each annotation includes the following terms: (a) A question; (b) One or more timestamp(s) on the question; (c) Four to five options; (d) A checkbox to pick the correct option","venue":null,"work_id":"f9539041-fab7-4205-9e3a-96573557c6f3","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:1dd862e5a51538140a5c6b4fcbfce51f0b7787ea0976809453f823e66e7cce04","observation_id":"64e7acfa-a56d-4779-ae9f-efec16703dbd","resolution":{"observed_at":"2026-05-18T00:30:12.512366Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"42d9f5c7-9306-4773-9962-248de501f4c5","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:ec2082934eef0f1d497f099045167ec16ce4571d69998ad109090174d3f6533e","observation_id":"f504f5b8-369a-4d64-826e-69f7fb354dd7","resolution":{"observed_at":"2026-05-18T00:30:12.515274Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"raw” data saved in addition to the preprocessed/cleaned/labeled data (e.g., to support unanticipated future uses)? If so, please provide a link or other access point to the “raw","venue":null,"work_id":"cdf1837d-36c1-4a12-a527-d51fb1cc683f","year":null},"citing_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-05-18T00:30:12.190492Z"},"links":{"citing_paper":"/paper/2407.15754"},"observation_digest":"sha256:73eb91ccc6213c4d3d346b04bc72ce22bd5c88cced0e9b96168e7ba637c01594","observation_id":"4cab70c6-e716-4d2f-9021-81c2b93ee0b0","resolution":{"observed_at":"2026-05-18T00:30:12.518723Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-06T14:12:05.548569Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding"},"reference_resolution":{"displayed":76,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":2,"unresolved":49,"verified_exact":1,"verified_fuzzy":24},"total_outbound_references":76},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 76 of 76 outbound references and 80 inbound Pith citation observations for arXiv:2407.15754."}