{"as_of":"2026-08-08T01:24:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:d3ca9b28de6b285538bdebff81b88cb056336e0937e2dc8942665d946d188f0c","coverage":[{"denominator":31,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":31,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:41:09.637593Z","state":"measured"},{"denominator":39,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":39,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":8,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":8,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T04:24:55.001360Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T00:07:28.700018Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.14321","snapshot_observed_at":"2026-08-03T13:01:57.956043Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2601.01095","last_updated":"2026-08-02T22:17:06Z","snapshot_observed_at":"2026-08-06T23:24:03.757272Z","submitted_at":"2026-01-03T07:12:55Z","title":"NarrativeTrack: Evaluating Entity-Centric Reasoning for Narrative Understanding","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-03T13:01:57.956043Z"},"links":{"cited_paper":"/paper/2505.14321","citing_paper":"/paper/2601.01095"},"observation_digest":"sha256:c22be1ebfac385b2b10e5b7887276a98ff1d86fa7d90893796c4ec92da8f67e4","observation_id":"18307915-d59b-4e62-944f-aa7c483f1624","resolution":{"observed_at":"2026-08-03T13:01:57.956043Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.14321","snapshot_observed_at":"2026-08-04T06:36:33.806787Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2601.01095","last_updated":"2026-08-02T22:17:06Z","snapshot_observed_at":"2026-08-06T23:24:03.757272Z","submitted_at":"2026-01-03T07:12:55Z","title":"NarrativeTrack: Evaluating Entity-Centric Reasoning for Narrative Understanding","version":4},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-04T06:36:33.806787Z"},"links":{"cited_paper":"/paper/2505.14321","citing_paper":"/paper/2601.01095"},"observation_digest":"sha256:3bd6c4002ecdc0eb2f50d2b6b90ef2019ac7859455b840bee01642f2c2944804","observation_id":"9afbff76-9efc-4df2-9eb1-4cf3663958cc","resolution":{"observed_at":"2026-08-04T06:36:33.806787Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"cited_work":{"arxiv_id":"2505.14321","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.14321","snapshot_observed_at":"2026-07-03T00:07:28.700018Z","title":"Breaking down video llm benchmarks: Knowledge, spatial perception, or true temporal understanding?","venue":null,"work_id":"86a2d8da-d610-44dd-aa77-36040be8ddf6","year":2025},"citing_paper":{"arxiv_id":"2604.15736","last_updated":"2026-04-17T06:22:41Z","snapshot_observed_at":"2026-07-06T23:03:12.000381Z","submitted_at":"2026-04-17T06:22:41Z","title":"RefereeBench: Are Video MLLMs Ready to be Multi-Sport Referees","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-10T08:38:27.081358Z"},"links":{"cited_paper":"/paper/2505.14321","citing_paper":"/paper/2604.15736"},"observation_digest":"sha256:463b67fb0eb7953f69ec7b944f1dd120408e4700cd184f1a9695087afcdb90ec","observation_id":"4af70d98-2788-4b20-85b2-abe90fc7600a","resolution":{"observed_at":"2026-05-10T08:48:02.252494Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"cited_work":{"arxiv_id":"2505.14321","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.14321","snapshot_observed_at":"2026-07-03T00:07:28.700018Z","title":"Breaking down video llm benchmarks: Knowledge, spatial perception, or true temporal understanding?","venue":null,"work_id":"86a2d8da-d610-44dd-aa77-36040be8ddf6","year":2025},"citing_paper":{"arxiv_id":"2605.01391","last_updated":"2026-06-11T07:06:15Z","snapshot_observed_at":"2026-07-06T23:14:38.379875Z","submitted_at":"2026-05-02T11:28:20Z","title":"VISTA: Video Interaction Spatio-Temporal Analysis Benchmark","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-09T15:18:00.436343Z"},"links":{"cited_paper":"/paper/2505.14321","citing_paper":"/paper/2605.01391"},"observation_digest":"sha256:753cb671a33757429ad9add06fda14b5ca0da40f81a2add4d4641260b6e763b2","observation_id":"8b721bc2-2993-4738-92b6-4fd426d49e1c","resolution":{"observed_at":"2026-05-11T16:41:22.517885Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"cited_work":{"arxiv_id":"2505.14321","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.14321","snapshot_observed_at":"2026-07-03T00:07:28.700018Z","title":"Breaking down video llm benchmarks: Knowledge, spatial perception, or true temporal understanding?","venue":null,"work_id":"86a2d8da-d610-44dd-aa77-36040be8ddf6","year":2025},"citing_paper":{"arxiv_id":"2605.01391","last_updated":"2026-06-11T07:06:15Z","snapshot_observed_at":"2026-07-06T23:14:38.379875Z","submitted_at":"2026-05-02T11:28:20Z","title":"VISTA: Video Interaction Spatio-Temporal Analysis Benchmark","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-07-01T00:24:25.203292Z"},"links":{"cited_paper":"/paper/2505.14321","citing_paper":"/paper/2605.01391"},"observation_digest":"sha256:7506e014568762bf26037f6bd5308c7a601449bcb77ea15bb7ae51495bb45a27","observation_id":"537c814d-9e7f-4b8f-819c-53b57e0df066","resolution":{"observed_at":"2026-07-01T00:25:09.422359Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"cited_work":{"arxiv_id":"2505.14321","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.14321","snapshot_observed_at":"2026-07-03T00:07:28.700018Z","title":"Breaking down video llm benchmarks: Knowledge, spatial perception, or true temporal understanding?","venue":null,"work_id":"86a2d8da-d610-44dd-aa77-36040be8ddf6","year":2025},"citing_paper":{"arxiv_id":"2606.00640","last_updated":"2026-05-30T09:30:30Z","snapshot_observed_at":"2026-08-06T02:39:29.306378Z","submitted_at":"2026-05-30T09:30:30Z","title":"An Attribute-Based Measure of Video Complexity","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-28T19:00:54.718177Z"},"links":{"cited_paper":"/paper/2505.14321","citing_paper":"/paper/2606.00640"},"observation_digest":"sha256:2f7c5fafd766d6f2bf513906b1ecac801ebf117675a262e06fdf0c4119f0b191","observation_id":"efdfaa89-8f69-4236-a9c6-9184a07e4464","resolution":{"observed_at":"2026-06-28T19:02:34.110328Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"cited_work":{"arxiv_id":"2505.14321","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.14321","snapshot_observed_at":"2026-07-03T00:07:28.700018Z","title":"Breaking down video llm benchmarks: Knowledge, spatial perception, or true temporal understanding?","venue":null,"work_id":"86a2d8da-d610-44dd-aa77-36040be8ddf6","year":2025},"citing_paper":{"arxiv_id":"2606.09646","last_updated":"2026-06-08T15:40:32Z","snapshot_observed_at":"2026-08-07T03:49:33.275871Z","submitted_at":"2026-06-08T15:40:32Z","title":"Do Video Foundation Models Understand Intuitive Physics? A Layerwise Probing Analysis","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-27T17:25:15.493925Z"},"links":{"cited_paper":"/paper/2505.14321","citing_paper":"/paper/2606.09646"},"observation_digest":"sha256:e59d4a7264983b8059567cfd5aaf53b407940a2f732ace34d31c8a8cec55729a","observation_id":"52eb70fb-bb63-4e0c-b641-84a3ea0666ea","resolution":{"observed_at":"2026-07-03T00:07:28.701935Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.14321","snapshot_observed_at":"2026-08-07T04:24:55.001360Z","title":"arXiv preprint arXiv:2505.14321 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.06361","last_updated":"2026-08-06T17:57:06Z","snapshot_observed_at":"2026-08-08T01:18:41.947571Z","submitted_at":"2026-08-06T17:57:06Z","title":"The Low Frequency Trap: Video Language Models Fail at Simple Event Bookkeeping","version":1},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-08-07T04:24:55.001360Z"},"links":{"cited_paper":"/paper/2505.14321","citing_paper":"/paper/2608.06361"},"observation_digest":"sha256:239386e7e19972f1988224d71fc09a0b4fa639bce28f1df4433ab6c756017c34","observation_id":"cd04825c-b05d-4ba5-9fad-add7deb39880","resolution":{"observed_at":"2026-08-07T04:24:55.001360Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2505.14321/citation-record","integrity":"/paper/2505.14321/integrity","json":"/paper/2505.14321/citation-record.json","paper":"/paper/2505.14321"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:41:06.956393Z","title":"Longvideobench: A benchmark for long-context interleaved video-language understanding, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:06.956393Z"},"links":{"citing_paper":"/paper/2505.14321"},"observation_digest":"sha256:146d474948739add6e4671b67451176406417e0fad2e82172bf96565c7d75167","observation_id":"4c33aa65-0e2e-42db-9161-a9511e63d799","resolution":{"observed_at":"2026-08-07T15:41:06.956393Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:41:12.249575Z","title":"Egoschema: A diagnostic benchmark for very long-form video language understanding.NeurIPS, 2024","venue":null,"work_id":"e54c5886-9e44-4c8f-99a4-5f8ccd46b6bf","year":2024},"citing_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:07.010477Z"},"links":{"citing_paper":"/paper/2505.14321"},"observation_digest":"sha256:29cf783c45ca0989014e25a551f1a0cd7223c9f28cb43bc2ebb9e2e9d1536cac","observation_id":"50720574-c593-4f33-843e-abb42ec207a6","resolution":{"observed_at":"2026-08-07T15:41:12.356042Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:41:11.960796Z","title":"NExT-QA: Next phase of question-answering to explaining temporal actions","venue":null,"work_id":"b6a0037d-5ca3-4f96-9358-3b0d9365b081","year":2021},"citing_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:07.083589Z"},"links":{"citing_paper":"/paper/2505.14321"},"observation_digest":"sha256:4bb310aaf4af4bfea1a191d9553bebd2f6c524ac0241c4d0e54c477fb2bf1c70","observation_id":"e5c37511-b97e-44fe-8da5-08490f695976","resolution":{"observed_at":"2026-08-07T15:41:12.088606Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.21075","last_updated":"2025-05-30T13:08:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-05-31T17:59:47Z","title":"Video-MME: The First-Ever Comprehensive Evaluation Benchmark of Multi-modal LLMs in Video Analysis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.21075","snapshot_observed_at":"2026-08-07T15:41:07.210366Z","title":"Video-mme: The first-ever comprehensive evaluation benchmark of multi-modal llms in video analysis.arXiv preprint arXiv:2405.21075, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:07.210366Z"},"links":{"cited_paper":"/paper/2405.21075","citing_paper":"/paper/2505.14321"},"observation_digest":"sha256:2e815d2d770210dcc6fb1b163dd5778e09e431365653a139edf93cba3469bbb8","observation_id":"d333cd39-fecf-4ff0-bd47-6f21b014e69d","resolution":{"observed_at":"2026-08-07T15:41:07.210366Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04264","last_updated":"2025-01-01T15:53:58Z","snapshot_observed_at":"2026-08-03T20:38:36.602554Z","submitted_at":"2024-06-06T17:09:32Z","title":"MLVU: Benchmarking Multi-task Long Video Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04264","snapshot_observed_at":"2026-08-07T15:41:07.306705Z","title":"Mlvu: A comprehensive benchmark for multi-task long video understanding.arXiv preprint arXiv:2406.04264, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:07.306705Z"},"links":{"cited_paper":"/paper/2406.04264","citing_paper":"/paper/2505.14321"},"observation_digest":"sha256:80ed452bfd14a3ed06b9be336907a63c81c4cd8b7c872ef222ea7973502501d0","observation_id":"37c1e67a-f6d3-49aa-bdf7-7ade136da3ed","resolution":{"observed_at":"2026-08-07T15:41:07.306705Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08035","last_updated":"2025-08-09T10:54:59Z","snapshot_observed_at":"2026-08-05T10:34:24.268925Z","submitted_at":"2024-06-12T09:36:52Z","title":"LVBench: An Extreme Long Video Understanding Benchmark","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08035","snapshot_observed_at":"2026-08-07T15:41:07.407582Z","title":"Lvbench: An extreme long video understanding benchmark.arXiv preprint arXiv:2406.08035, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:07.407582Z"},"links":{"cited_paper":"/paper/2406.08035","citing_paper":"/paper/2505.14321"},"observation_digest":"sha256:0d77679af82d82693ac364e153a83ec96f0010eaf0376ecb42288d07351b918c","observation_id":"1747ad97-fddd-4dfb-b21d-bbf26f516b93","resolution":{"observed_at":"2026-08-07T15:41:07.407582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:41:11.712560Z","title":"Perception test: A diagnostic benchmark for multimodal video models.Advances in Neural Information Processing Systems, 36:42748– 42761, 2023","venue":null,"work_id":"20fd6fbc-3df9-48f9-bef2-3eac7e8591aa","year":2023},"citing_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:07.469778Z"},"links":{"citing_paper":"/paper/2505.14321"},"observation_digest":"sha256:ddc686f9c1f04261a20911f46df77d0434b1e49afd70608d4368138913ebb5f7","observation_id":"f3f18659-5bfb-4b0e-a23b-f1fb93941091","resolution":{"observed_at":"2026-08-07T15:41:11.817503Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:41:11.427677Z","title":"Palm: Scaling language modeling with pathways.JMLR, 2023","venue":null,"work_id":"9da58d50-4b12-41bb-bcd7-8be5090bb739","year":2023},"citing_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:07.584674Z"},"links":{"citing_paper":"/paper/2505.14321"},"observation_digest":"sha256:cf1f5b66e8bb2ceb132422b2b86e4bd5857e1c0bfb49e5dfb4980a7339fae4c8","observation_id":"ff29bc6e-46c5-44b8-b2d6-8873122790c4","resolution":{"observed_at":"2026-08-07T15:41:11.568082Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-07T15:41:07.658408Z","title":"Llama 2: Open foundation and fine-tuned chat models.arXiv:2307.09288, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:07.658408Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2505.14321"},"observation_digest":"sha256:d82f54829d456810f1feeb72491d9d27f07785eed905c3b252d2113df511315a","observation_id":"eedf7de9-9ee6-4b19-a0b1-7a1e74ef8153","resolution":{"observed_at":"2026-08-07T15:41:07.658408Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-07T15:41:07.716508Z","title":"GPT-4 technical report","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:07.716508Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2505.14321"},"observation_digest":"sha256:8615f9e022ea9536645371b459440b4891cc66f3c9d36fea22205dc3164849b6","observation_id":"9860e96a-faf4-4a6f-b7e1-5c126b77a85f","resolution":{"observed_at":"2026-08-07T15:41:07.716508Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03744","last_updated":"2024-05-15T19:22:44Z","snapshot_observed_at":"2026-07-06T16:28:22.350574Z","submitted_at":"2023-10-05T17:59:56Z","title":"Improved Baselines with Visual Instruction Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03744","snapshot_observed_at":"2026-08-07T15:41:07.780818Z","title":"Improved baselines with visual instruction tuning.arXiv:2310.03744, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:07.780818Z"},"links":{"cited_paper":"/paper/2310.03744","citing_paper":"/paper/2505.14321"},"observation_digest":"sha256:dbc30052af34750e0ab7eb43654b07974d2cb49177dd248694bb412aa882334c","observation_id":"05209c6a-ebe1-463a-a8dd-5a27d1eb520d","resolution":{"observed_at":"2026-08-07T15:41:07.780818Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:41:11.191165Z","title":"LLaV A- NeXT: Improved reasoning, ocr, and world knowledge, 2024","venue":null,"work_id":"64fb978e-752c-4e3b-8bdd-1a0ede443bc2","year":2024},"citing_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:07.876391Z"},"links":{"citing_paper":"/paper/2505.14321"},"observation_digest":"sha256:65fd0f5e81dd120102e359f495e68cd3f7e8a69609ea3a74da7765ab442938b8","observation_id":"1f978819-19ad-46ba-927d-11dabe00f256","resolution":{"observed_at":"2026-08-07T15:41:11.296636Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08024","last_updated":"2024-06-12T09:22:45Z","snapshot_observed_at":"2026-08-06T04:30:23.299517Z","submitted_at":"2024-06-12T09:22:45Z","title":"Fewer Tokens and Fewer Videos: Extending Video Understanding Abilities in Large Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2406.08024","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.08024","snapshot_observed_at":"2026-08-07T15:41:09.929010Z","title":"Fewer Tokens and Fewer Videos: Extending Video Understanding Abilities in Large Vision-Language Models","venue":"cs.CV","work_id":"0a6d8c66-db97-435f-b4bf-a5e5f1a5ca8d","year":2024},"citing_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:07.984130Z"},"links":{"cited_paper":"/paper/2406.08024","citing_paper":"/paper/2505.14321"},"observation_digest":"sha256:4af1c514eec9067276b702c44d803815cf9dc1f0e0f6439faa61ff0ae055d432","observation_id":"1db472a9-81e2-4a50-8841-9cefcd3006ee","resolution":{"observed_at":"2026-08-07T15:41:09.987240Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:41:10.982826Z","title":"Video-ChatGPT: Towards detailed video understanding via large vision and language models","venue":null,"work_id":"7d53736a-6991-4851-9b53-0b9cc9ab05b2","year":2024},"citing_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:08.081997Z"},"links":{"citing_paper":"/paper/2505.14321"},"observation_digest":"sha256:5d49aa7bc70dc6839f18b270417ee007fdad9174276a06631574dadff5314e54","observation_id":"d3ce082d-2d69-45ce-a096-ac5d33a01150","resolution":{"observed_at":"2026-08-07T15:41:11.091970Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:41:08.147677Z","title":"Learning transferable visual models from natural language supervision","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:08.147677Z"},"links":{"citing_paper":"/paper/2505.14321"},"observation_digest":"sha256:d3ce5f2910f2e2df559aaa4ccb035a561dfe81bb4b808474083b02ed6a08031a","observation_id":"d045ce46-38f7-49f5-a98b-3709d6d129a6","resolution":{"observed_at":"2026-08-07T15:41:08.147677Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:41:10.746758Z","title":"LLaV A-NeXT: A strong zero-shot video understanding model, 2024","venue":null,"work_id":"f0415827-6fe0-4bd2-afbe-7334fa92e7fe","year":2024},"citing_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:08.205977Z"},"links":{"citing_paper":"/paper/2505.14321"},"observation_digest":"sha256:c386aa092624f1098a5a2de442a5694a3cd9c24d758339c7c37c570c75db01df","observation_id":"77a759a2-488b-4a4f-812a-0eb18daddc67","resolution":{"observed_at":"2026-08-07T15:41:10.817989Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01071","last_updated":"2025-08-02T13:32:09Z","snapshot_observed_at":"2026-07-06T19:09:10.244642Z","submitted_at":"2024-09-02T08:52:58Z","title":"VideoLLaMB: Long Streaming Video Understanding with Recurrent Memory Bridges","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01071","snapshot_observed_at":"2026-08-07T15:41:08.331967Z","title":"Videollamb: Long-context video understanding with recurrent memory bridges.arXiv:2409.01071, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:08.331967Z"},"links":{"cited_paper":"/paper/2409.01071","citing_paper":"/paper/2505.14321"},"observation_digest":"sha256:53e11d1cf5b2f7a9d576a930a493aacb5d6dcbc1704fa2341fac22f71026cec1","observation_id":"ae24c0ee-71d4-42d7-b8f4-2ced1bbe7b05","resolution":{"observed_at":"2026-08-07T15:41:08.331967Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-07T15:41:08.446932Z","title":"Internvideo2","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:08.446932Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2505.14321"},"observation_digest":"sha256:216d15575a5a09422e1ca7d3c4a665bd3f91e2d0de2de229cda55e5356eab0fc","observation_id":"33c74869-a7ea-414b-91da-4bde70336a44","resolution":{"observed_at":"2026-08-07T15:41:08.446932Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.10360","last_updated":"2024-12-13T18:53:24Z","snapshot_observed_at":"2026-08-06T08:59:28.938249Z","submitted_at":"2024-12-13T18:53:24Z","title":"Apollo: An Exploration of Video Understanding in Large Multimodal Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.10360","snapshot_observed_at":"2026-08-07T15:41:08.568366Z","title":"Apollo: An exploration of video understanding in large multimodal models.arXiv:2412.10360, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:08.568366Z"},"links":{"cited_paper":"/paper/2412.10360","citing_paper":"/paper/2505.14321"},"observation_digest":"sha256:0468d37f774724681a0bb2505acfa4db918dced44b707d513047a1a2c2fd61d5","observation_id":"73db5c87-2b89-4b6e-abd9-9da8866b837f","resolution":{"observed_at":"2026-08-07T15:41:08.568366Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:41:10.545996Z","title":"Oryx MLLM: On-demand spatial-temporal understanding at arbitrary resolution.ICLR, 2025","venue":null,"work_id":"936be503-4323-46e6-89cb-1665130e71b9","year":2025},"citing_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:08.690765Z"},"links":{"citing_paper":"/paper/2505.14321"},"observation_digest":"sha256:673012f20f3188218fac80b04af7665b0f2dea92546e86fa6ad93f45e3399b61","observation_id":"15403d8c-9c3f-4dc4-87cf-6aab6ed4073b","resolution":{"observed_at":"2026-08-07T15:41:10.631715Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13106","last_updated":"2025-06-03T03:33:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T18:59:46Z","title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13106","snapshot_observed_at":"2026-08-07T15:41:08.771392Z","title":"VideoLLaMA 3: Frontier multimodal foundation models for image and video understanding.arXiv:2501.13106, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:08.771392Z"},"links":{"cited_paper":"/paper/2501.13106","citing_paper":"/paper/2505.14321"},"observation_digest":"sha256:3c6b00796eeae130fdda436f7aade2c675be03b19d6cfe42575c881f6317ce5b","observation_id":"1ce74f51-9a08-4df3-8d8f-bffae76d590a","resolution":{"observed_at":"2026-08-07T15:41:08.771392Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:41:10.377632Z","title":"Video question answering via gradually refined attention over appearance and motion","venue":null,"work_id":"af74d050-399e-4074-bdd0-084e0446d346","year":2017},"citing_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:08.863605Z"},"links":{"citing_paper":"/paper/2505.14321"},"observation_digest":"sha256:cdeac032bd0eaa8d827608d3ae8b495abf8d922e384373a83380fc77fdf57d0e","observation_id":"18674d71-7b64-49eb-a2e6-63ad3c5caffa","resolution":{"observed_at":"2026-08-07T15:41:10.460870Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:41:10.201018Z","title":"ActivityNet-QA: A dataset for understanding complex web videos via question answering","venue":null,"work_id":"82119ce4-2e9d-41f5-9f5c-1bcbd191da16","year":2019},"citing_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:08.944154Z"},"links":{"citing_paper":"/paper/2505.14321"},"observation_digest":"sha256:b54a9c34fd12288f65f59a38d7dbbbe6bbb1de3bb6bf59e985160f6dce3ae500","observation_id":"7b72c423-7680-4d6e-a3c0-1b7b844ae2b1","resolution":{"observed_at":"2026-08-07T15:41:10.295260Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:41:09.035873Z","title":"Mvbench: A comprehensive multi-modal video understanding benchmark","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:09.035873Z"},"links":{"citing_paper":"/paper/2505.14321"},"observation_digest":"sha256:6fce1d4640a1442b0bde9a56c8f36a52b3a4c89c4bdb95b63e9654d97b14fa62","observation_id":"8f4aaf61-ea0e-4a96-bc14-66fe79aab981","resolution":{"observed_at":"2026-08-07T15:41:09.035873Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.06851","last_updated":"2024-10-13T18:11:26Z","snapshot_observed_at":"2026-07-06T19:13:28.897964Z","submitted_at":"2024-09-10T20:19:14Z","title":"LIME: Less Is More for MLLM Evaluation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.06851","snapshot_observed_at":"2026-08-07T15:41:09.118503Z","title":"Lime: Less is more for mllm evaluation.arXiv preprint arXiv:2409.06851, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:09.118503Z"},"links":{"cited_paper":"/paper/2409.06851","citing_paper":"/paper/2505.14321"},"observation_digest":"sha256:b6f77c6b3d8b8f45cd39205d33f6fe8f692fbe58291dfb0521dc2cf57d26ceb2","observation_id":"13727959-433c-4784-a7c7-20bf3e898152","resolution":{"observed_at":"2026-08-07T15:41:09.118503Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15841","last_updated":"2024-09-15T05:00:18Z","snapshot_observed_at":"2026-07-06T18:50:13.559866Z","submitted_at":"2024-07-22T17:58:04Z","title":"SlowFast-LLaVA: A Strong Training-Free Baseline for Video Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15841","snapshot_observed_at":"2026-08-07T15:41:09.214825Z","title":"SlowFast-LLaV A: A strong training-free baseline for video large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:09.214825Z"},"links":{"cited_paper":"/paper/2407.15841","citing_paper":"/paper/2505.14321"},"observation_digest":"sha256:3879ed533e494b526c69343b3652c93a848154e9b57ce75001c6a8d7582e413a","observation_id":"7a46c902-f8b5-4033-88b7-c49360ce7c9a","resolution":{"observed_at":"2026-08-07T15:41:09.214825Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16994","last_updated":"2024-04-29T14:52:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-25T19:29:55Z","title":"PLLaVA : Parameter-free LLaVA Extension from Images to Videos for Video Dense Captioning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16994","snapshot_observed_at":"2026-08-07T15:41:09.310216Z","title":"PLLaV A: Parameter-free llava extension from images to videos for video dense captioning.arXiv:2404.16994, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:09.310216Z"},"links":{"cited_paper":"/paper/2404.16994","citing_paper":"/paper/2505.14321"},"observation_digest":"sha256:b964768059dd784b448a1f7dd788343e489e550489bf23eb68654a874cbefb0e","observation_id":"8d3ba9d7-f312-453b-8ac2-1aca202aa967","resolution":{"observed_at":"2026-08-07T15:41:09.310216Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03326","last_updated":"2024-10-26T16:35:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:59:44Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03326","snapshot_observed_at":"2026-08-07T15:41:09.367899Z","title":"Llava-onevision: Easy visual task transfer.arXiv preprint arXiv:2408.03326, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:09.367899Z"},"links":{"cited_paper":"/paper/2408.03326","citing_paper":"/paper/2505.14321"},"observation_digest":"sha256:580d2d805a7b6ed5f07d8c1bc2b73870ada7fe8125a293aa5144db78e360cf73","observation_id":"f1591ffc-cd7f-4534-bac7-83080ef6bccc","resolution":{"observed_at":"2026-08-07T15:41:09.367899Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05530","last_updated":"2024-12-16T17:39:39Z","snapshot_observed_at":"2026-07-06T17:41:42.995949Z","submitted_at":"2024-03-08T18:54:20Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05530","snapshot_observed_at":"2026-08-07T15:41:09.462250Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context.arXiv preprint arXiv:2403.05530, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:09.462250Z"},"links":{"cited_paper":"/paper/2403.05530","citing_paper":"/paper/2505.14321"},"observation_digest":"sha256:4726be8afe055cd260afb8fc3eaacec6794e11b90829fdce7dd96ec9f4607234","observation_id":"d0d88534-9569-4b16-a312-cc1126394219","resolution":{"observed_at":"2026-08-07T15:41:09.462250Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12191","last_updated":"2024-10-03T15:54:49Z","snapshot_observed_at":"2026-08-06T05:35:29.109022Z","submitted_at":"2024-09-18T17:59:32Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12191","snapshot_observed_at":"2026-08-07T15:41:09.539358Z","title":"Qwen2-vl: Enhancing vision-language model’s perception of the world at any resolution.arXiv preprint arXiv:2409.12191, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:09.539358Z"},"links":{"cited_paper":"/paper/2409.12191","citing_paper":"/paper/2505.14321"},"observation_digest":"sha256:e9ebc75e38a7be57ef672c7851663eaa5be2a76ee34fcef9db876b9bbb4b679b","observation_id":"cd5c3eea-f837-4aaa-9500-c77784b901eb","resolution":{"observed_at":"2026-08-07T15:41:09.539358Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.10122","last_updated":"2024-10-01T12:07:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-16T10:59:44Z","title":"Video-LLaVA: Learning United Visual Representation by Alignment Before Projection","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.10122","snapshot_observed_at":"2026-08-07T15:41:09.637593Z","title":"Video-LLaV A: Learning united visual representation by alignment before projection.arXiv:2311.10122, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:09.637593Z"},"links":{"cited_paper":"/paper/2311.10122","citing_paper":"/paper/2505.14321"},"observation_digest":"sha256:7defc9bfd929536096ce159497d74a685572fed45f3fdcf0935f1c50658be363","observation_id":"9223dad6-8fbc-4ace-ba0a-386a2e5366f4","resolution":{"observed_at":"2026-08-07T15:41:09.637593Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2505.14321","last_updated":"2025-05-20T13:07:55Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-07T23:44:06.171904Z","submitted_at":"2025-05-20T13:07:55Z","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?"},"reference_resolution":{"displayed":31,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":20,"verified_exact":1,"verified_fuzzy":10},"total_outbound_references":31},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 31 of 31 outbound references and 8 inbound Pith citation observations for arXiv:2505.14321."}