{"as_of":"2026-08-07T18:01:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e46b2131ab21e8ed589ce1ec296b9ca490e2ad2ed11c1f635f942b05c18d0d26","coverage":[{"denominator":58,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":58,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T15:48:38.262186Z","state":"measured"},{"denominator":59,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":59,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T09:12:09.932226Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.15028","snapshot_observed_at":"2026-08-04T09:12:09.932226Z","title":"Towards video thinking test: A holistic benchmark for advanced video reasoning and understanding","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2510.17045","last_updated":"2026-06-01T07:19:41Z","snapshot_observed_at":"2026-08-04T09:12:02.562884Z","submitted_at":"2025-10-19T23:17:13Z","title":"Video Reasoning without Training","version":2},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-04T09:12:09.932226Z"},"links":{"cited_paper":"/paper/2507.15028","citing_paper":"/paper/2510.17045"},"observation_digest":"sha256:0011b3a8ce1afca40c76d6ca8e7b6cdf7144028bd9ceaa6ccc3fd84774ad8605","observation_id":"029bc2f4-af1c-4e0d-9ee6-6c811843f413","resolution":{"observed_at":"2026-08-04T09:12:09.932226Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2507.15028/citation-record","integrity":"/paper/2507.15028/integrity","json":"/paper/2507.15028/citation-record.json","paper":"/paper/2507.15028"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:47.934876Z","title":"Vqa: Visual question answering","venue":null,"work_id":"968a5bce-7a9c-4288-965f-574f83ebec7c","year":2015},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:30.939545Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:ea339f4d55f73427d6e17302fcc6db022d119abe33efb29f82a75aa586a5b706","observation_id":"5df95d99-6457-4549-98e8-bc2296fb4eaf","resolution":{"observed_at":"2026-08-06T15:48:48.049637Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.10818","last_updated":"2024-10-15T17:55:46Z","snapshot_observed_at":"2026-07-06T19:33:16.949210Z","submitted_at":"2024-10-14T17:59:58Z","title":"TemporalBench: Benchmarking Fine-grained Temporal Understanding for Multimodal Video Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.10818","snapshot_observed_at":"2026-08-06T15:48:31.035101Z","title":"Temporalbench: Towards fine-grained temporal understanding for multimodal video models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:31.035101Z"},"links":{"cited_paper":"/paper/2410.10818","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:080eb3d895ec9e2f65647c11530a0de381321dbacb9a6c7589206e2433b54a4d","observation_id":"f91d00e8-c6f6-491f-b1a5-7bdc970597a8","resolution":{"observed_at":"2026-08-06T15:48:31.035101Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:47.679540Z","title":"Collecting highly paral- lel data for paraphrase evaluation","venue":null,"work_id":"48a6013b-eb83-441c-b3af-1a3dedbbf4f2","year":2011},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:31.146121Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:7d74e420b25f16ac5cd3ffed0d8429e721fc84b963ca48fc8b24220fe9e52037","observation_id":"455bff05-8a2e-4bca-bcb1-2426ad781cbc","resolution":{"observed_at":"2026-08-06T15:48:47.771747Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.05271","last_updated":"2025-09-26T12:52:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-06T18:57:08Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.05271","snapshot_observed_at":"2026-08-06T15:48:31.309034Z","title":"Expanding performance boundaries of open-source multimodal models with model, data, and test- time scaling","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:31.309034Z"},"links":{"cited_paper":"/paper/2412.05271","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:31b8140f12a0cc14a6c15a00f46ffd0f9209d27f57d3e0d83027537345b5e19c","observation_id":"5d1d99ad-54ad-465f-83df-4725f87d8d78","resolution":{"observed_at":"2026-08-06T15:48:31.309034Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14432","last_updated":"2025-05-02T16:03:31Z","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:59:55Z","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.14432","snapshot_observed_at":"2026-08-06T15:48:31.423977Z","title":"Insight-v: Ex- ploring long-chain visual reasoning with multimodal large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:31.423977Z"},"links":{"cited_paper":"/paper/2411.14432","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:d7fde377aec9773f52672f178ed2257d0abf0dec3dc924a710f09bb3238946b6","observation_id":"7811dae3-8c3b-4a69-ac98-5634d2a7ae5b","resolution":{"observed_at":"2026-08-06T15:48:31.423977Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14515","last_updated":"2024-10-30T13:38:10Z","snapshot_observed_at":"2026-07-06T18:34:24.078145Z","submitted_at":"2024-06-20T17:26:01Z","title":"MMBench-Video: A Long-Form Multi-Shot Benchmark for Holistic Video Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.14515","snapshot_observed_at":"2026-08-06T15:48:31.569319Z","title":"Mmbench-video: A long-form multi-shot benchmark for holistic video under- standing","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:31.569319Z"},"links":{"cited_paper":"/paper/2406.14515","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:c71d9da35b652292e7590b6a2da23973ae8ce16279914293ca4adef9f7804d81","observation_id":"419a10bf-2fc6-4e6e-a497-af875e6cbe22","resolution":{"observed_at":"2026-08-06T15:48:31.569319Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.21075","last_updated":"2025-05-30T13:08:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-05-31T17:59:47Z","title":"Video-MME: The First-Ever Comprehensive Evaluation Benchmark of Multi-modal LLMs in Video Analysis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.21075","snapshot_observed_at":"2026-08-06T15:48:31.692811Z","title":"Video-mme: The first-ever compre- hensive evaluation benchmark of multi-modal llms in video analysis","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:31.692811Z"},"links":{"cited_paper":"/paper/2405.21075","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:5c4d03330906c9c44911a47b42b1c70946ae33e42efbf660d9ff8d97c9e2c2a0","observation_id":"20cae88d-8699-4879-bab5-a606e491c550","resolution":{"observed_at":"2026-08-06T15:48:31.692811Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:47.480091Z","title":"Agqa: A benchmark for compositional spatio-temporal reasoning","venue":null,"work_id":"c2b6018d-d546-47c4-8a90-c8a9e9dc28d7","year":2021},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:31.798160Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:c1ffb3020d2bbb550bac3ddc71cad9ebd32a81d484fc87aa7abd8e978e5a8e8f","observation_id":"6601ce9b-8936-4466-a2e9-2e1660cd6698","resolution":{"observed_at":"2026-08-06T15:48:47.570794Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:47.165096Z","title":"Similarity and fea- tures of natural textures","venue":null,"work_id":"8e66a362-bf86-400f-ace1-f974dbe9c716","year":1999},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:31.967725Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:9dc087edbe8cb7dd034b6739a81331d9259323aace23fb926d694836bb1f5baf","observation_id":"c5bb15e9-d108-40c0-95af-fc0fd2632a22","resolution":{"observed_at":"2026-08-06T15:48:47.327426Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:46.894405Z","title":"Natural adversarial examples","venue":null,"work_id":"12a70ea2-0d52-4a08-90b0-5a45d8e46dd0","year":2021},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:32.140486Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:8e505a41a6be6c1228745ee9c7f53c3869a74e228cd38cb37f76e186f20ef83b","observation_id":"fef48a7e-f210-40f0-a400-b5f99936a1a2","resolution":{"observed_at":"2026-08-06T15:48:47.012478Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:46.672298Z","title":"Video-mmmu: Evaluating knowledge acquisition from multi-discipline pro- fessional videos, 2025","venue":null,"work_id":"8c162250-3475-4007-8048-2a7a2a036942","year":2025},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:32.242853Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:034df44844a656260ed72e3b1059b6d5df1df4837e85d17ce858a5e5a7f043cf","observation_id":"db349333-1200-42ed-9571-7d04be0ba812","resolution":{"observed_at":"2026-08-06T15:48:46.740041Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:32.351024Z","title":"Tgif-qa: Toward spatio-temporal reasoning in visual question answering","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:32.351024Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:4eb5402bef40d189d9ae328aee951c0b91473e02ea14935ce63d57754fa8810e","observation_id":"09e9c5cb-554a-4e44-a014-9f9206135d35","resolution":{"observed_at":"2026-08-06T15:48:32.351024Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:46.399398Z","title":"Robust modeling in cognitive science","venue":null,"work_id":"67ab8535-41f0-496c-b77b-1408b46ec1e2","year":2019},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:32.475759Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:60f53d4cbb87138f104f4d78178d41f19da8c019566c782c297c89679fda357c","observation_id":"ddf0b73b-170f-460c-a1b1-b11ec693ce2a","resolution":{"observed_at":"2026-08-06T15:48:46.500872Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1809.01696","last_updated":"2019-05-07T21:34:05Z","snapshot_observed_at":"2026-07-06T06:59:26.080263Z","submitted_at":"2018-09-05T19:14:11Z","title":"TVQA: Localized, Compositional Video Question Answering","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1809.01696","snapshot_observed_at":"2026-08-06T15:48:32.606188Z","title":"Tvqa: Localized, compositional video question answering","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:32.606188Z"},"links":{"cited_paper":"/paper/1809.01696","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:99e31067914336c64b4623425eede789f08a6b04824322cef371542ff20dc2f9","observation_id":"3a2e6533-99d6-4ee3-b0d6-3e5259281859","resolution":{"observed_at":"2026-08-06T15:48:32.606188Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:46.139646Z","title":"Mvbench: A comprehensive multi- modal video understanding benchmark, 2023","venue":null,"work_id":"1b6d864c-04d9-46fc-84cf-17e8df53e309","year":2023},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:32.768765Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:9af6df09df3a6c8d39abacff87face150c807ca6a1c0a6e4a5d9571fbf15ef15","observation_id":"1ac2132a-7884-43c0-b223-c3ae4ba09978","resolution":{"observed_at":"2026-08-06T15:48:46.290809Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:32.871605Z","title":"Grounded language-image pre-training","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:32.871605Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:e3f219fcbf77ea4966bcef07fa5ca917e51f8d579c34b97faf0ff563a269eda5","observation_id":"a15c7e92-a2ef-4d72-87a3-cf47fee2a286","resolution":{"observed_at":"2026-08-06T15:48:32.871605Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00476","last_updated":"2024-06-03T04:13:39Z","snapshot_observed_at":"2026-08-04T21:17:37.211833Z","submitted_at":"2024-03-01T12:02:19Z","title":"TempCompass: Do Video LLMs Really Understand Videos?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00476","snapshot_observed_at":"2026-08-06T15:48:33.000709Z","title":"Tempcom- pass: Do video llms really understand videos?arXiv preprint arXiv:2403.00476, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:33.000709Z"},"links":{"cited_paper":"/paper/2403.00476","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:b3adb75d2fdb4bd322be87ffc000edc8185f7e269d74204c1426927783c882f5","observation_id":"7196bd3a-3cfa-4d87-8e80-86f386fdbdef","resolution":{"observed_at":"2026-08-06T15:48:33.000709Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12961","last_updated":"2025-02-27T06:09:46Z","snapshot_observed_at":"2026-07-06T19:18:17.523057Z","submitted_at":"2024-09-19T17:59:51Z","title":"Oryx MLLM: On-Demand Spatial-Temporal Understanding at Arbitrary Resolution","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12961","snapshot_observed_at":"2026-08-06T15:48:33.180292Z","title":"Oryx mllm: On-demand spatial-temporal understanding at arbitrary resolution","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:33.180292Z"},"links":{"cited_paper":"/paper/2409.12961","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:6de62139ea2d88c4d5673ee44552dbbc03b233e3992b88110ce6d901d98b24b1","observation_id":"ef22287e-6b45-409f-83b5-5da365626f32","resolution":{"observed_at":"2026-08-06T15:48:33.180292Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12966","last_updated":"2024-03-21T16:26:44Z","snapshot_observed_at":"2026-07-06T17:47:13.814205Z","submitted_at":"2024-03-19T17:59:52Z","title":"Chain-of-Spot: Interactive Reasoning Improves Large Vision-Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12966","snapshot_observed_at":"2026-08-06T15:48:33.294475Z","title":"Chain-of-spot: Interactive reasoning improves large vision-language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:33.294475Z"},"links":{"cited_paper":"/paper/2403.12966","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:9736d0ba7596a1a3fa68dc38d4125cb417c4afad526f2058789b1efe7e5778cf","observation_id":"c23e29bd-49d4-4ee2-a256-716b3e5727d9","resolution":{"observed_at":"2026-08-06T15:48:33.294475Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.04328","last_updated":"2025-06-02T19:33:24Z","snapshot_observed_at":"2026-07-06T20:32:23.502480Z","submitted_at":"2025-02-06T18:59:55Z","title":"Ola: Pushing the Frontiers of Omni-Modal Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.04328","snapshot_observed_at":"2026-08-06T15:48:33.446078Z","title":"Ola: Pushing the frontiers of omni-modal language model with progressive modality alignment","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:33.446078Z"},"links":{"cited_paper":"/paper/2502.04328","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:f4a5bd39f9a9760aef76c2699cb323bde730efe0cbef385aff8099ade37a485a","observation_id":"37c6c68c-dfa0-458c-aa6f-d438fb959ff9","resolution":{"observed_at":"2026-08-06T15:48:33.446078Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:45.886243Z","title":"Video detail caption, 2024","venue":null,"work_id":"143ea81d-f66b-4324-8262-4d86298f4f5e","year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:33.588967Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:f0356a12426d16cb3132c34c6f3b79719d297bfa1435f8419584cacf4a3e84b5","observation_id":"765b1288-3e8a-46f3-a236-2e205096e505","resolution":{"observed_at":"2026-08-06T15:48:46.002685Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:45.675868Z","title":"Video-chatgpt: Towards detailed video understanding via large vision and language models","venue":null,"work_id":"fd2b8a8b-b14f-4ec7-ba3a-fa78c7be67ee","year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:33.712835Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:6303c0d770f1729c3cd5e55a937fb39b2757b2988030d8caac9265294055da5e","observation_id":"d89edf69-5883-4e24-b7c9-6462a3481db4","resolution":{"observed_at":"2026-08-06T15:48:45.768516Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:45.433037Z","title":"Egoschema: A diagnostic benchmark for very long- form video language understanding","venue":null,"work_id":"6ff2c6c7-a62a-471a-97e1-325c110eaf81","year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:33.843680Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:e3b58f6233a63ed248446ca067c96e6e3aca4b3a8a2830b89c8cf428225b4d8f","observation_id":"a8110d16-6f0f-469c-847f-5dddec64fb52","resolution":{"observed_at":"2026-08-06T15:48:45.555971Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:45.155189Z","title":"Identifying the perceptual dimensions of visual complexity of scenes","venue":null,"work_id":"9d4b4eb4-0089-4f74-9ae2-2ea3949808e8","year":2004},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:33.963318Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:ded8e541400eb2c6b64331895096b43204e8e67443d98291b60705f31c9b5cb5","observation_id":"e2f1aa5c-f826-4ed0-bcbd-fdbc2fef3663","resolution":{"observed_at":"2026-08-06T15:48:45.268685Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:44.843980Z","title":"Hello gpt-4o","venue":null,"work_id":"58f93450-5d5c-413e-af85-9991954f50a6","year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:34.090197Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:716aa09fb5177f51f39e9028899c5665977585fccbec7fe9b3bbb3c50250a38a","observation_id":"48f1ad09-e144-43d9-8af3-4afa5c535bfe","resolution":{"observed_at":"2026-08-06T15:48:44.992721Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:44.552561Z","title":"Robustness analysis of video- language models against visual and language perturbations","venue":null,"work_id":"4809e0df-b226-4e37-841f-a645d3b65212","year":2022},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:34.215109Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:a1210ea8e8c8f874b5a63337c0ebb6745270738d1c6d9b2122b2a879f6d1e6af","observation_id":"8b509cc9-209d-490f-bf22-4e17d8a20f18","resolution":{"observed_at":"2026-08-06T15:48:44.690772Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:44.332717Z","title":"Visual cot: Advancing multi-modal language models with a com- prehensive dataset and benchmark for chain-of-thought rea- soning","venue":null,"work_id":"a5bfdfc1-0b2c-4d15-81bd-63609b355e41","year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:34.327788Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:d2ad95902f5825fef88aaa1dac5bb503cb7d8cefa139cadd8e6cc71ba78a62e3","observation_id":"8624ad6a-af68-4a15-b742-e3fedcfbcd4c","resolution":{"observed_at":"2026-08-06T15:48:44.432636Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:44.107487Z","title":"Complex narratives","venue":null,"work_id":"802da6c0-7adc-4164-afca-a268675e3240","year":2014},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:34.436761Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:ec7cd69d3290ec48b2dbfe85f39f742188d1064aa5f70423053d412101e4a517","observation_id":"3c072a10-ee74-403c-baa0-46613aa21e3a","resolution":{"observed_at":"2026-08-06T15:48:44.211901Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:43.871086Z","title":"A standardized set of 260 pictures: norms for name agreement, image agree- ment, familiarity, and visual complexity","venue":null,"work_id":"7fac0123-6001-4c81-a4b7-a1e9564a4663","year":null},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:34.545309Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:0cb287d13a0c2c73f84b334b0b67dc9ed0965492a6865ff5450ee4c899e082b8","observation_id":"8b651d54-b4a8-49bb-baa6-c740cfbb7d51","resolution":{"observed_at":"2026-08-06T15:48:43.994154Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.08862","last_updated":"2025-02-10T02:00:10Z","snapshot_observed_at":"2026-08-05T06:17:39.997811Z","submitted_at":"2024-08-16T17:44:02Z","title":"Visual Agents as Fast and Slow Thinkers","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.08862","snapshot_observed_at":"2026-08-06T15:48:34.682053Z","title":"Visual agents as fast and slow thinkers","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:34.682053Z"},"links":{"cited_paper":"/paper/2408.08862","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:86f5e3fd3baea01fbf57b09947e825f00a4b6819196a630dd045c81c3b81defa","observation_id":"83a81e53-1273-43de-940c-88950fadb301","resolution":{"observed_at":"2026-08-06T15:48:34.682053Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:43.615916Z","title":"Curious objects: How vi- sual complexity guides attention and engagement","venue":null,"work_id":"ad1a61fa-e208-4d0c-98e8-a219710b7b8b","year":2021},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:34.828072Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:68191636ec40255fea1ce829fb4f9d4ba5f75dd693f524c761c3bcc69b9795ce","observation_id":"4be120cf-78ac-448b-a092-a75e3fc7285c","resolution":{"observed_at":"2026-08-06T15:48:43.757236Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:43.409073Z","title":"Cognitive load during problem solving: Ef- fects on learning","venue":null,"work_id":"66565297-ed5d-4103-b0c4-c2879f44bcb4","year":1988},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:34.956258Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:5b274a046dcebc3244dba81188433b9ab360223290339ed974aa530e0df206b9","observation_id":"c6d3932e-6194-486a-946a-42cb352693bb","resolution":{"observed_at":"2026-08-06T15:48:43.510219Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-06T15:48:35.106376Z","title":"Gemini: a family of highly capable multimodal models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:35.106376Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:820c1fe0995ad9378305099ab9703abbc63c2975542d4345cd795f74a38417e9","observation_id":"80026ae7-918d-488a-a749-558296988b73","resolution":{"observed_at":"2026-08-06T15:48:35.106376Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:43.186330Z","title":"Qwen2.5-vl, 2025","venue":null,"work_id":"6622a7a4-c06f-4e35-8063-be20a79dc686","year":2025},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:35.204531Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:e7e99e8faad4366ed307c00abdec1a0c9ff4b9b9c6549e558bfd011257e9fe70","observation_id":"42b5846c-c080-4b1c-b886-c596ca59958e","resolution":{"observed_at":"2026-08-06T15:48:43.298939Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:42.956743Z","title":"Chain-of-thought prompting elicits reasoning in large language models, 2023","venue":null,"work_id":"0de39dbc-3803-4c7d-8d5b-02a3e0f4d7fc","year":2023},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:35.335517Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:485de2f3f036d99c9e2274bf76590645b9e1a0d1872d0e971135f5eceb6cb6d7","observation_id":"fc779f9f-e966-485d-b1a5-df5446d54495","resolution":{"observed_at":"2026-08-06T15:48:43.068789Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:42.744992Z","title":"Star: A benchmark for situated reasoning in real-world videos","venue":null,"work_id":"015c6eb0-ca80-4bae-8b7a-d1fc760aa07b","year":2021},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:35.438170Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:f109b52e3c92438bdc5f7d2409bc4746b93682367b87cd6cfd33d32631cb72e6","observation_id":"544bbc8a-a175-4be3-93a3-8ee3c3b255c1","resolution":{"observed_at":"2026-08-06T15:48:42.832074Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:42.437368Z","title":"Longvideobench: A benchmark for long-context interleaved video-language understanding, 2024","venue":null,"work_id":"3e2a4b94-d942-4e41-aedf-e6341a8f1e53","year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:35.549580Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:fda1d9f19745ae0be41bfa842394c189e101af1403c3325778d46f3fae75d7fb","observation_id":"d1b49f1d-6949-4034-b8e3-6b30f8aca8ca","resolution":{"observed_at":"2026-08-06T15:48:42.558730Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:42.216183Z","title":"Next-qa: Next phase of question-answering to explaining temporal actions","venue":null,"work_id":"b6fbeae4-3c78-43fb-8ede-63258147079b","year":2021},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:35.677619Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:27dd545960d315e8948a98fca28999fc45b25190316c3784ce07744d6bbed419","observation_id":"b79f48c1-7e3f-451e-8075-04db7d05888d","resolution":{"observed_at":"2026-08-06T15:48:42.321877Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14899","last_updated":"2024-03-22T13:24:35Z","snapshot_observed_at":"2026-07-31T06:40:35.464819Z","submitted_at":"2023-06-26T17:59:55Z","title":"FunQA: Towards Surprising Video Comprehension","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.14899","snapshot_observed_at":"2026-08-06T15:48:35.898141Z","title":"Funqa: Towards surprising video comprehension","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:35.898141Z"},"links":{"cited_paper":"/paper/2306.14899","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:b23ab8f7dd5b476bb679fc95cfd6dccfc9963d3e16c2f8a99f9aa875385d9a0b","observation_id":"be4cf59e-3032-4c5d-bc8b-fb4d9ffe5508","resolution":{"observed_at":"2026-08-06T15:48:35.898141Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:41.908306Z","title":"Video question answer- ing via gradually refined attention over appearance and mo- tion","venue":null,"work_id":"dadc6706-3013-49a2-9090-ff9675cf5b23","year":2017},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:36.010706Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:74583abc5968612a4130ca0b34c1ed2aadd137c277e607f1622bc4383aa12fd7","observation_id":"8615b8f0-4b17-4cfd-8b16-7733e1e6d649","resolution":{"observed_at":"2026-08-06T15:48:42.030420Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:41.636169Z","title":"Sutd-trafficqa: A question answering benchmark and an efficient network for video rea- soning over traffic events","venue":null,"work_id":"9c4dec88-49a3-4418-b4aa-b518d7a37b3b","year":2021},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:36.149550Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:27c15e19c81763700f16e87fd9b9db3a1c3a361f3aef5132789bc5201b62dc91","observation_id":"e765a4ed-3fc3-4c43-a5a5-09eb60da7363","resolution":{"observed_at":"2026-08-06T15:48:41.727394Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1910.01442","last_updated":"2020-03-08T00:09:07Z","snapshot_observed_at":"2026-07-06T08:26:38.349660Z","submitted_at":"2019-10-03T13:16:36Z","title":"CLEVRER: CoLlision Events for Video REpresentation and Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1910.01442","snapshot_observed_at":"2026-08-06T15:48:36.257872Z","title":"Clevrer: Collision events for video representation and reasoning","venue":null,"work_id":null,"year":1910},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:36.257872Z"},"links":{"cited_paper":"/paper/1910.01442","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:e0d9a3e3eecd2c54ca5a7969ab2a98ff0545d8963b573d9370cfaa080522b411","observation_id":"698ac212-0b18-4618-aebe-1225e15ab204","resolution":{"observed_at":"2026-08-06T15:48:36.257872Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:41.418479Z","title":"Activitynet-qa: A dataset for understanding complex web videos via question answering","venue":null,"work_id":"1ec97e30-39e9-4d29-bb5b-e84dfe02f0f7","year":2019},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:36.361773Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:0d9ecc2cc4605f14c355d7d16c5567205362c0b523d6217ff07ced0270c1b0f3","observation_id":"b02e17e6-0383-4c89-b9c7-9ef1a00d10c9","resolution":{"observed_at":"2026-08-06T15:48:41.517617Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:41.162839Z","title":"Activitynet-qa: A dataset for understanding complex web videos via question answering","venue":null,"work_id":"663a5957-adad-4089-8885-a14c2efea13d","year":2019},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:36.499719Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:b3cb28a49d32e133ed163dff854159901e3cee00ad7a74e34333a72cafcfd0f6","observation_id":"b49b54d2-abcb-4a86-aee5-8c7353405cd7","resolution":{"observed_at":"2026-08-06T15:48:41.286050Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:40.893094Z","title":"Social-iq: A question answer- ing benchmark for artificial social intelligence","venue":null,"work_id":"1af5b41b-ae24-4c93-9334-cea56cab7464","year":2019},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:36.656128Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:64d83d959fc2c145470b1e3cef444f205d74a5645c8dd9c042baa8d87e121aa5","observation_id":"29c5a716-8008-4753-9afe-d0c154f4e367","resolution":{"observed_at":"2026-08-06T15:48:41.022509Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.02858","last_updated":"2023-10-25T06:23:31Z","snapshot_observed_at":"2026-07-06T15:38:39.712379Z","submitted_at":"2023-06-05T13:17:27Z","title":"Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.02858","snapshot_observed_at":"2026-08-06T15:48:36.756457Z","title":"Video-llama: An instruction-tuned audio-visual language model for video un- derstanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:36.756457Z"},"links":{"cited_paper":"/paper/2306.02858","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:f76cc957f807d007ac2d748c6af6dfa5bbdf03a63066a6990121005fc9a9a7e3","observation_id":"263fd12e-d9a0-49d4-8027-6217a5f54e3e","resolution":{"observed_at":"2026-08-06T15:48:36.756457Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:40.621301Z","title":"B- avibench: Towards evaluating the robustness of large vision- language model on black-box adversarial visual-instructions,","venue":null,"work_id":"225b70f2-3394-4894-a640-91fb863f1529","year":null},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:36.911831Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:5f2668b95496963fa76c0bdb2e03bde5d4e5183f96bd41497677217180ece861","observation_id":"bbe8c10f-0b64-438a-b966-196f252f44a8","resolution":{"observed_at":"2026-08-06T15:48:40.773699Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:37.032657Z","title":"Lmms- eval: Reality check on the evaluation of large multimodal models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:37.032657Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:3f534d35d5dfc0152f221edd5c8732bb709eb82fc584fda1b83df143637bb8e3","observation_id":"7d7965d5-2b29-4fac-8610-7ec7b3bbff7a","resolution":{"observed_at":"2026-08-06T15:48:37.032657Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.16852","last_updated":"2024-07-01T02:59:29Z","snapshot_observed_at":"2026-08-07T09:52:45.942315Z","submitted_at":"2024-06-24T17:58:06Z","title":"Long Context Transfer from Language to Vision","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.16852","snapshot_observed_at":"2026-08-06T15:48:37.124321Z","title":"Long context transfer from language to vision","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:37.124321Z"},"links":{"cited_paper":"/paper/2406.16852","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:6dda6abf04e4d46b3628ceb0211611d420ad9c20175b56d3b1ba820a2595550c","observation_id":"fe22ba31-6ef0-422e-a89e-4b5188af112e","resolution":{"observed_at":"2026-08-06T15:48:37.124321Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:40.352479Z","title":"Llava- next: A strong zero-shot video understanding model, 2024","venue":null,"work_id":"f9273919-402d-4881-aa20-2386964149b7","year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:37.232691Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:691261250eb22f53ea5649d7b5914f42af8428d1f2bb003de29a83e75fac74ed","observation_id":"0d4b1904-9a86-4848-a923-e83820d76d7e","resolution":{"observed_at":"2026-08-06T15:48:40.466201Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:40.108946Z","title":"Video instruction tuning with synthetic data, 2024","venue":null,"work_id":"582c093d-8b28-4262-8dd6-6f2982998f7a","year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:37.388074Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:099f5973382d3152f071daa0eaa453cefe6e16a5f23f4c2e7e44e4bbb3e5713b","observation_id":"8364d8e5-e494-420c-88ef-7445affc3200","resolution":{"observed_at":"2026-08-06T15:48:40.216826Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:39.857946Z","title":"Worldqa: Multimodal world knowledge in videos through long-chain reasoning, 2024","venue":null,"work_id":"41ee5d78-ad52-438e-92ed-0c4c3b2f2299","year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:37.519206Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:8a2de32a6f0a415056ecfcd5a625be0de5ccffb05ebd1447abf1a83586cba496","observation_id":"91e9c0ea-54b4-49e1-91ac-05896924ac87","resolution":{"observed_at":"2026-08-06T15:48:39.995188Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04264","last_updated":"2025-01-01T15:53:58Z","snapshot_observed_at":"2026-08-03T20:38:36.602554Z","submitted_at":"2024-06-06T17:09:32Z","title":"MLVU: Benchmarking Multi-task Long Video Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04264","snapshot_observed_at":"2026-08-06T15:48:37.637372Z","title":"Mlvu: A comprehensive benchmark for multi-task long video understanding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:37.637372Z"},"links":{"cited_paper":"/paper/2406.04264","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:91ae0b4722baba855ebc4faee1d93ee01056f3754d8b8750b2a1092b5d42373a","observation_id":"2571e13e-793f-4fc5-bb6d-3ebd87c5ccdd","resolution":{"observed_at":"2026-08-06T15:48:37.637372Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:39.645273Z","title":"Hierarchical video content description and summarization using unified semantic and visual similarity","venue":null,"work_id":"d6457f4f-efae-442a-bc6a-3638934b16cf","year":2003},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:37.755434Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:a4d3d19038ff972323ab8c72fdc976108410bd3793d88ad23b92821509737710","observation_id":"ddccc800-85a9-4126-838a-ac6ec14d2461","resolution":{"observed_at":"2026-08-06T15:48:39.748619Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:39.331297Z","title":"In total, the annotation process cost 8227.32 human hours","venue":null,"work_id":"88293fa2-49fc-4913-a82b-bc64bf2e33ba","year":null},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:37.878023Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:7a939a887bf12548c8d1f58f21047d3adff58ab9f07c4528b6833cb7421064b7","observation_id":"12f08e3e-6f6d-4d24-8a18-9afa369badf2","resolution":{"observed_at":"2026-08-06T15:48:39.494867Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:39.081973Z","title":"• Aparaphrased correct be the set of videos where the para- phrased open-ended question is answered correctly","venue":null,"work_id":"70b7dde5-dcf9-42cd-a411-9a78d35ccd59","year":null},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:37.984186Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:df57519085c648a9e63ec81248d46b0dc291bc31ef2c4d1d15faceee1ac1e01e","observation_id":"5e1f3085-8a05-494a-8fe5-7f5260b953d0","resolution":{"observed_at":"2026-08-06T15:48:39.209624Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:38.862420Z","title":"3 shows the prompt for evaluating open-ended an- swers","venue":null,"work_id":"71f834d3-6d97-49fd-9599-db830df97928","year":null},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:38.104212Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:31c6f4df08fdd65f997ce7827b190c8e46c2fc63bc53032a63d22f65b7a0a198","observation_id":"4661c413-e575-40cf-8cb5-b6b4260aa7d2","resolution":{"observed_at":"2026-08-06T15:48:38.958699Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:38.649973Z","title":"element” and “event","venue":null,"work_id":"cdbe8f52-3a99-49a9-acfa-3d4fbb222535","year":null},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:38.262186Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:a20bd0d84edcabbe6336c0be6e973c154f5edf93cbc905257ef7f3528861bdd7","observation_id":"f2b4833d-0cf6-4494-a4ee-ad5b033b6135","resolution":{"observed_at":"2026-08-06T15:48:38.751950Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-06T20:42:17.108788Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding"},"reference_resolution":{"displayed":58,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":20,"verified_exact":0,"verified_fuzzy":38},"total_outbound_references":58},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 58 of 58 outbound references and 1 inbound Pith citation observation for arXiv:2507.15028."}