{"as_of":"2026-08-08T17:57:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:1c239df3acdc71dce4db3af036ef7f6985f8f9c9091c91763d318ecf2514d3e0","coverage":[{"denominator":55,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":55,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:34:44.884301Z","state":"measured"},{"denominator":65,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":65,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":10,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":10,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T12:50:02.846860Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-02T19:07:17.426534Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.14640","snapshot_observed_at":"2026-08-07T12:50:02.846860Z","title":"Videoeval-pro: Robust and realistic long video understanding evaluation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.03179","last_updated":"2025-05-29T13:17:25Z","snapshot_observed_at":"2026-08-08T00:26:20.360231Z","submitted_at":"2025-05-29T13:17:25Z","title":"Vid-SME: Membership Inference Attacks against Large Video Understanding Models","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T12:50:02.846860Z"},"links":{"cited_paper":"/paper/2505.14640","citing_paper":"/paper/2506.03179"},"observation_digest":"sha256:d1ecaeeb3ad772ce07f827ec448163c69318871c2a73fddd7ea87ff68a8b26ab","observation_id":"a0e8be73-74c8-40ed-ac7e-74a5cf6f213b","resolution":{"observed_at":"2026-08-07T12:50:02.846860Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.14640","snapshot_observed_at":"2026-08-06T23:13:11.304314Z","title":"Videoeval-pro: Robust and realistic long video understanding evaluation.arXiv preprint arXiv:2505.14640, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.19225","last_updated":"2025-06-24T01:19:56Z","snapshot_observed_at":"2026-08-07T21:07:46.308380Z","submitted_at":"2025-06-24T01:19:56Z","title":"Video-XL-2: Towards Very Long-Video Understanding Through Task-Aware KV Sparsification","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T23:13:11.304314Z"},"links":{"cited_paper":"/paper/2505.14640","citing_paper":"/paper/2506.19225"},"observation_digest":"sha256:d1d02b72ad653bf12590c4a3a0da4455fc825efd664aac78ce6e03ef39031ea4","observation_id":"c7fc560d-b42f-45f3-a437-4be1ade2a1d1","resolution":{"observed_at":"2026-08-06T23:13:11.304314Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"cited_work":{"arxiv_id":"2505.14640","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.14640","snapshot_observed_at":"2026-07-02T19:07:17.426534Z","title":"Videoeval-pro: Robust and realistic long video understanding evaluation","venue":null,"work_id":"25cc74dd-a972-4446-b09b-7f85d9171e13","year":2025},"citing_paper":{"arxiv_id":"2601.10611","last_updated":"2026-04-02T16:01:02Z","snapshot_observed_at":"2026-08-02T06:45:30.387180Z","submitted_at":"2026-01-15T17:27:44Z","title":"Molmo2: Open Weights and Data for Vision-Language Models with Video Understanding and Grounding","version":4},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-05-16T04:21:29.526008Z"},"links":{"cited_paper":"/paper/2505.14640","citing_paper":"/paper/2601.10611"},"observation_digest":"sha256:38baeb34ceae4b6564f5b12ca50c4eebc99266bf1aa8d24589f08d6b93c0baa9","observation_id":"39eb555c-eb3a-4d13-b80d-f9c4fac7983e","resolution":{"observed_at":"2026-05-16T04:21:29.796098Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"cited_work":{"arxiv_id":"2505.14640","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.14640","snapshot_observed_at":"2026-07-02T19:07:17.426534Z","title":"Videoeval-pro: Robust and realistic long video understanding evaluation","venue":null,"work_id":"25cc74dd-a972-4446-b09b-7f85d9171e13","year":2025},"citing_paper":{"arxiv_id":"2602.22779","last_updated":"2026-06-03T09:11:47Z","snapshot_observed_at":"2026-08-02T20:38:33.149530Z","submitted_at":"2026-02-26T09:15:34Z","title":"TrajTok: Learning Trajectory Tokens enables better Video Understanding","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-15T19:11:52.694778Z"},"links":{"cited_paper":"/paper/2505.14640","citing_paper":"/paper/2602.22779"},"observation_digest":"sha256:20a98871a6281b64e1e7fc5463f4deb483254726fb70723d1d159450d1187a41","observation_id":"47adbc6b-9538-4066-9e0a-a2ccd64bb11c","resolution":{"observed_at":"2026-05-15T19:16:31.811999Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.14640","snapshot_observed_at":"2026-08-02T20:38:40.952686Z","title":"Videoeval-pro: Robust and realistic long video understanding evaluation.arXiv preprint arXiv:2505.14640, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.22779","last_updated":"2026-06-03T09:11:47Z","snapshot_observed_at":"2026-08-02T20:38:33.149530Z","submitted_at":"2026-02-26T09:15:34Z","title":"TrajTok: Learning Trajectory Tokens enables better Video Understanding","version":3},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-02T20:38:40.952686Z"},"links":{"cited_paper":"/paper/2505.14640","citing_paper":"/paper/2602.22779"},"observation_digest":"sha256:d22562a43de40a0b49f8ada8ce35f4ecffd166e43a22aacf555e444e43d4320f","observation_id":"920533a0-f3fe-4684-a31e-1a4b0b8223d9","resolution":{"observed_at":"2026-08-02T20:38:40.952686Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.14640","snapshot_observed_at":"2026-07-13T15:38:41.390945Z","title":"arXiv preprint arXiv:2505.14640 (2025) 2","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.29616","last_updated":"2026-07-02T03:18:44Z","snapshot_observed_at":"2026-08-07T03:57:41.978126Z","submitted_at":"2026-03-31T11:37:42Z","title":"Video-Oasis: Rethinking Evaluation of Video Understanding","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-07-13T15:38:41.390945Z"},"links":{"cited_paper":"/paper/2505.14640","citing_paper":"/paper/2603.29616"},"observation_digest":"sha256:15b710867d0e37ff0ac420b06c4486c949c03688d16f4a6fd586edf97d51dbd8","observation_id":"13235850-057d-420d-b4cf-99c338778fd2","resolution":{"observed_at":"2026-07-13T15:38:41.390945Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"cited_work":{"arxiv_id":"2505.14640","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.14640","snapshot_observed_at":"2026-07-02T19:07:17.426534Z","title":"Videoeval-pro: Robust and realistic long video understanding evaluation","venue":null,"work_id":"25cc74dd-a972-4446-b09b-7f85d9171e13","year":2025},"citing_paper":{"arxiv_id":"2605.10434","last_updated":"2026-05-11T12:06:57Z","snapshot_observed_at":"2026-07-06T23:22:23.287792Z","submitted_at":"2026-05-11T12:06:57Z","title":"WorldReasonBench: Human-Aligned Stress Testing of Video Generators as Future World-State Predictors","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-12T03:39:01.254916Z"},"links":{"cited_paper":"/paper/2505.14640","citing_paper":"/paper/2605.10434"},"observation_digest":"sha256:a92bb92120c3800c4d630c5d91a6783112ee905dc089af0056766ddece0cda74","observation_id":"4b5eeacb-ba77-4cfc-97f9-0dba3b9c2840","resolution":{"observed_at":"2026-05-12T07:11:25.138117Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"cited_work":{"arxiv_id":"2505.14640","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.14640","snapshot_observed_at":"2026-07-02T19:07:17.426534Z","title":"Videoeval-pro: Robust and realistic long video understanding evaluation","venue":null,"work_id":"25cc74dd-a972-4446-b09b-7f85d9171e13","year":2025},"citing_paper":{"arxiv_id":"2607.00248","last_updated":"2026-06-30T22:57:43Z","snapshot_observed_at":"2026-08-03T15:00:12.134373Z","submitted_at":"2026-06-30T22:57:43Z","title":"Seed2.0 Model Card: Towards Intelligence Frontier for Real-World Complexity","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-07-02T18:57:46.841456Z"},"links":{"cited_paper":"/paper/2505.14640","citing_paper":"/paper/2607.00248"},"observation_digest":"sha256:b5fbc8bb0799d7b71e988ba4cea573601751f82bcea6a722389599048d2d09c9","observation_id":"280e7169-00b3-4883-bce5-1fc4884e04b7","resolution":{"observed_at":"2026-07-02T19:07:17.427951Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.14640","snapshot_observed_at":"2026-08-02T00:44:46.835407Z","title":"Videoeval-pro: Robust and realistic long video understanding evaluation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14935","last_updated":"2026-07-16T12:47:59Z","snapshot_observed_at":"2026-08-06T19:13:04.526298Z","submitted_at":"2026-07-16T12:47:59Z","title":"VideoChat3: Fully Open Video MLLM for Efficient and Generalist Video Understanding","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-02T00:44:46.835407Z"},"links":{"cited_paper":"/paper/2505.14640","citing_paper":"/paper/2607.14935"},"observation_digest":"sha256:d4af34e0bb239f8267ef076d18925b1f4cc13b809336446bc8c5b6c2440343f5","observation_id":"84ee4e48-c29d-4129-aa2e-0182b72b4e4e","resolution":{"observed_at":"2026-08-02T00:44:46.835407Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.14640","snapshot_observed_at":"2026-07-31T06:20:14.256778Z","title":"VideoEval-Pro: Robust and realistic long video understanding evaluation.arXiv preprint arXiv:2505.14640, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.24904","last_updated":"2026-07-27T17:59:53Z","snapshot_observed_at":"2026-08-07T06:47:24.083796Z","submitted_at":"2026-07-27T17:59:53Z","title":"Mage-VL: An Efficient Codec-Native Streaming Multimodal Foundation Model","version":1},"reference_index":155,"source":"pdf_text","source_observed_at":"2026-07-31T06:20:14.256778Z"},"links":{"cited_paper":"/paper/2505.14640","citing_paper":"/paper/2607.24904"},"observation_digest":"sha256:fc145643c343ec9e097b8e47c23f16d05f917b6cd333e02f3f85c3fd3a49aa43","observation_id":"40587f60-b205-4071-bfe4-bd9aa258a220","resolution":{"observed_at":"2026-07-31T06:20:14.256778Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2505.14640/citation-record","integrity":"/paper/2505.14640/integrity","json":"/paper/2505.14640/citation-record.json","paper":"/paper/2505.14640"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2406.08035","last_updated":"2025-08-09T10:54:59Z","snapshot_observed_at":"2026-08-05T10:34:24.268925Z","submitted_at":"2024-06-12T09:36:52Z","title":"LVBench: An Extreme Long Video Understanding Benchmark","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08035","snapshot_observed_at":"2026-08-07T15:34:40.208786Z","title":"Lvbench: An extreme long video understanding benchmark","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:40.208786Z"},"links":{"cited_paper":"/paper/2406.08035","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:a967750bfe88f706e7d7af311751ef7c0821cb8982ce91440a30ae994c1302b9","observation_id":"6a88210c-b53b-4908-b4e2-4d404068d158","resolution":{"observed_at":"2026-08-07T15:34:40.208786Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.21075","last_updated":"2025-05-30T13:08:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-05-31T17:59:47Z","title":"Video-MME: The First-Ever Comprehensive Evaluation Benchmark of Multi-modal LLMs in Video Analysis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.21075","snapshot_observed_at":"2026-08-07T15:34:40.317680Z","title":"Video-mme: The first-ever comprehensive evaluation benchmark of multi-modal llms in video analysis.arXiv preprint arXiv:2405.21075, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:40.317680Z"},"links":{"cited_paper":"/paper/2405.21075","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:1a5f7b7aca98d8c7c25ec367217e27acd17cdbfa31c958d27d91c44019122eec","observation_id":"10ed969e-dfad-49c1-9133-7e0750fa635e","resolution":{"observed_at":"2026-08-07T15:34:40.317680Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05702","last_updated":"2024-01-11T07:09:44Z","snapshot_observed_at":"2026-07-06T17:14:07.183954Z","submitted_at":"2024-01-11T07:09:44Z","title":"Video Anomaly Detection and Explanation via Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05702","snapshot_observed_at":"2026-08-07T15:34:40.477549Z","title":"Video anomaly detection and explanation via large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:40.477549Z"},"links":{"cited_paper":"/paper/2401.05702","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:5cad907fcd3bdf16ff3e2b1cd6cf87432b3d9277f5b3a5e9ef32567617ecf5b2","observation_id":"1e7ad542-9a07-4b88-bba4-3cd40cdc2b28","resolution":{"observed_at":"2026-08-07T15:34:40.477549Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:34:40.586160Z","title":"Large scale interactive motion forecasting for autonomous driving: The waymo open motion dataset","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:40.586160Z"},"links":{"citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:f166461650134a20cf66aeb3df6df1396f0e77f0b8fecf7eba43ac16912b0cd6","observation_id":"9d5f0c6b-03c9-4506-9af8-f0464c479552","resolution":{"observed_at":"2026-08-07T15:34:40.586160Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:34:40.675590Z","title":"Towards automatic learning of procedures from web instructional videos","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:40.675590Z"},"links":{"citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:79aed327b21099a15b16c48a78fc4cf96ceefaf2e396b1387f64ce47839b2f4f","observation_id":"e77b6f11-b337-4d0a-9a24-bfdfca92ecb7","resolution":{"observed_at":"2026-08-07T15:34:40.675590Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.16852","last_updated":"2024-07-01T02:59:29Z","snapshot_observed_at":"2026-08-07T09:52:45.942315Z","submitted_at":"2024-06-24T17:58:06Z","title":"Long Context Transfer from Language to Vision","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.16852","snapshot_observed_at":"2026-08-07T15:34:40.769437Z","title":"Long context transfer from language to vision","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:40.769437Z"},"links":{"cited_paper":"/paper/2406.16852","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:fcaa8fa2b456358cea4d74d9222f091e1adfb2d3090d21f97a1617b181290c89","observation_id":"fbb17e75-ffea-475a-947c-10ab364bd87b","resolution":{"observed_at":"2026-08-07T15:34:40.769437Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.10188","last_updated":"2024-12-13T02:32:06Z","snapshot_observed_at":"2026-08-05T14:57:53.592979Z","submitted_at":"2024-08-19T17:48:08Z","title":"LongVILA: Scaling Long-Context Visual Language Models for Long Videos","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.10188","snapshot_observed_at":"2026-08-07T15:34:40.848837Z","title":"Longvila: Scaling long-context visual language models for long videos.arXiv preprint arXiv:2408.10188, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:40.848837Z"},"links":{"cited_paper":"/paper/2408.10188","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:11430d0dadc09e174df625c926941468c19559090d91dc081704a72387425340","observation_id":"f52197b2-3d0b-4a6e-8276-e907aa5a7935","resolution":{"observed_at":"2026-08-07T15:34:40.848837Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.00574","last_updated":"2025-07-13T16:21:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-31T18:01:23Z","title":"VideoChat-Flash: Hierarchical Compression for Long-Context Video Modeling","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.00574","snapshot_observed_at":"2026-08-07T15:34:40.968075Z","title":"Videochat-flash: Hierarchical compression for long-context video modeling.arXiv preprint arXiv:2501.00574, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:40.968075Z"},"links":{"cited_paper":"/paper/2501.00574","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:68e33e57ab1711bac40361116615eea2efb63dfe68f57e81609997e2fba47b72","observation_id":"0e55e230-5270-4ea2-ba9d-10a36c347294","resolution":{"observed_at":"2026-08-07T15:34:40.968075Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12386","snapshot_observed_at":"2026-08-07T15:34:41.079838Z","title":"Internvideo2","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:41.079838Z"},"links":{"cited_paper":"/paper/2501.12386","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:4d46265f395425fbcdbe791d80f8f1d01b9f19e66cb6c514ff69876ebdf90064","observation_id":"a97207d3-7a08-4b04-b7be-0cd06df75772","resolution":{"observed_at":"2026-08-07T15:34:41.079838Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.11579","last_updated":"2025-07-16T11:39:44Z","snapshot_observed_at":"2026-08-07T17:03:23.976828Z","submitted_at":"2025-03-14T16:45:23Z","title":"Vamba: Understanding Hour-Long Videos with Hybrid Mamba-Transformers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.11579","snapshot_observed_at":"2026-08-07T15:34:41.182215Z","title":"Vamba: Under- standing hour-long videos with hybrid mamba-transformers.arXiv preprint arXiv:2503.11579, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:41.182215Z"},"links":{"cited_paper":"/paper/2503.11579","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:aabc7f8acc34554ba246fda2f559e2ff836af5368e7a3e7d53aed8bae166f1af","observation_id":"08eef1c9-d3c6-4b8e-a355-ced9c4eaa905","resolution":{"observed_at":"2026-08-07T15:34:41.182215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:34:41.295575Z","title":"Token-efficient long video understanding for multimodal llms.arXiv preprint arXiv:2503.04130, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:41.295575Z"},"links":{"citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:13a45adbf4036f07aa3067e285147a5df038c4f98c86429ce7f910ab10ea7b1d","observation_id":"809ec41f-47d5-4412-a93c-ca8d346ecae4","resolution":{"observed_at":"2026-08-07T15:34:41.295575Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09590","last_updated":"2025-03-13T17:14:31Z","snapshot_observed_at":"2026-08-07T17:09:25.628961Z","submitted_at":"2025-03-12T17:57:32Z","title":"BIMBA: Selective-Scan Compression for Long-Range Video Question Answering","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09590","snapshot_observed_at":"2026-08-07T15:34:41.423639Z","title":"Bimba: Selective-scan compression for long-range video question answering.arXiv preprint arXiv:2503.09590, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:41.423639Z"},"links":{"cited_paper":"/paper/2503.09590","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:2a370a65b1291268346c7cae1a3202848c48dcd5e5f041a2aadb4c999acce0dd","observation_id":"3ac6fee9-ec53-406f-b2b4-f7c2e83cff95","resolution":{"observed_at":"2026-08-07T15:34:41.423639Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T15:34:41.495843Z","title":"Video instruction tuning with synthetic data.arXiv preprint arXiv:2410.02713, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:41.495843Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:00e1313d4cf39399506f3cce636bb84c2761298ce443b890aa1f00a96dddde03","observation_id":"aece33d9-7349-46da-b737-66bdd2dbf49b","resolution":{"observed_at":"2026-08-07T15:34:41.495843Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.00927","last_updated":"2024-12-01T18:27:28Z","snapshot_observed_at":"2026-08-08T11:33:10.278036Z","submitted_at":"2024-12-01T18:27:28Z","title":"VISTA: Enhancing Long-Duration and High-Resolution Video Understanding by Video Spatiotemporal Augmentation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.00927","snapshot_observed_at":"2026-08-07T15:34:41.595618Z","title":"Vista: Enhancing long- duration and high-resolution video understanding by video spatiotemporal augmentation.arXiv preprint arXiv:2412.00927, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:41.595618Z"},"links":{"cited_paper":"/paper/2412.00927","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:fe23fa3ff9e5b191ab66798918b71e7d45203f26a733facbe1f2063d8f62831c","observation_id":"c716fa61-58fb-4624-89ce-fe750b5d2a98","resolution":{"observed_at":"2026-08-07T15:34:41.595618Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.06958","last_updated":"2025-11-11T08:30:00Z","snapshot_observed_at":"2026-08-02T02:31:33.589341Z","submitted_at":"2025-04-09T15:09:27Z","title":"VideoChat-R1: Enhancing Spatio-Temporal Perception via Reinforcement Fine-Tuning","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.06958","snapshot_observed_at":"2026-08-07T15:34:41.669333Z","title":"Videochat-r1: Enhancing spatio-temporal perception via reinforce- ment fine-tuning.arXiv preprint arXiv:2504.06958, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:41.669333Z"},"links":{"cited_paper":"/paper/2504.06958","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:571d9bbf1aa0f6aff28b34a6785fac4389d6c08b24b5a7ab73e530d7863035fc","observation_id":"34216a65-fbda-4284-9c4c-1b6854591650","resolution":{"observed_at":"2026-08-07T15:34:41.669333Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.21776","last_updated":"2025-10-22T16:42:24Z","snapshot_observed_at":"2026-08-05T07:15:29.998948Z","submitted_at":"2025-03-27T17:59:51Z","title":"Video-R1: Reinforcing Video Reasoning in MLLMs","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.21776","snapshot_observed_at":"2026-08-07T15:34:41.743812Z","title":"Video-r1: Reinforcing video reasoning in mllms.arXiv preprint arXiv:2503.21776, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:41.743812Z"},"links":{"cited_paper":"/paper/2503.21776","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:ce2689b8df3b0bf4efcd9561b360e4ebf19ea18b9afab7fb59037766ff2d6b21","observation_id":"157d2750-e5aa-4ea4-9f87-c7e1e62a4782","resolution":{"observed_at":"2026-08-07T15:34:41.743812Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.10122","last_updated":"2024-10-01T12:07:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-16T10:59:44Z","title":"Video-LLaVA: Learning United Visual Representation by Alignment Before Projection","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.10122","snapshot_observed_at":"2026-08-07T15:34:41.886033Z","title":"Video-llava: Learning united visual representation by alignment before projection.arXiv preprint arXiv:2311.10122, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:41.886033Z"},"links":{"cited_paper":"/paper/2311.10122","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:2acdd5bac86545c43bbb63b7b04cffdb77e585a57f625ea59a07f54a1a31c5e0","observation_id":"4f4b19fd-09e3-4e4d-95d7-a4aaf5e427f6","resolution":{"observed_at":"2026-08-07T15:34:41.886033Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.18478","last_updated":"2025-04-27T10:39:51Z","snapshot_observed_at":"2026-08-07T16:41:53.661756Z","submitted_at":"2025-03-24T09:21:48Z","title":"Video-XL-Pro: Reconstructive Token Compression for Extremely Long Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.18478","snapshot_observed_at":"2026-08-07T15:34:41.992087Z","title":"Video-xl-pro: Re- constructive token compression for extremely long video understanding.arXiv preprint arXiv:2503.18478, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:41.992087Z"},"links":{"cited_paper":"/paper/2503.18478","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:9d53609f9bc99606aea78e71de9287d21daf4ffd89ce138fd78722b8b136f98e","observation_id":"3e33eeb5-d633-4f8a-b2ed-99519e4ae733","resolution":{"observed_at":"2026-08-07T15:34:41.992087Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04264","last_updated":"2025-01-01T15:53:58Z","snapshot_observed_at":"2026-08-03T20:38:36.602554Z","submitted_at":"2024-06-06T17:09:32Z","title":"MLVU: Benchmarking Multi-task Long Video Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04264","snapshot_observed_at":"2026-08-07T15:34:42.108152Z","title":"Mlvu: A comprehensive benchmark for multi-task long video understanding.arXiv preprint arXiv:2406.04264, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:42.108152Z"},"links":{"cited_paper":"/paper/2406.04264","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:c77ad72843db1b35333d6b8d3f2d72d7ac73bdd3ba47cef76bc3e0258b1650a9","observation_id":"5d257b92-d56b-4914-867c-72506d73fc7e","resolution":{"observed_at":"2026-08-07T15:34:42.108152Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:34:42.202827Z","title":"Longvideobench: A benchmark for long- context interleaved video-language understanding.Advances in Neural Information Processing Systems, 37:28828–28857, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:42.202827Z"},"links":{"citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:2c47f548f6e946f26b08395824cda0fc4aa7b94dd559f4d8e0b88b62834199d7","observation_id":"c0f84637-01cb-4874-a219-6fbf78ea032e","resolution":{"observed_at":"2026-08-07T15:34:42.202827Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:34:46.199161Z","title":"Hourvideo: 1-hour video- language understanding.Advances in Neural Information Processing Systems, 37:53168–53197, 2024","venue":null,"work_id":"e03ef94e-285e-48dd-895e-1047290dd7ba","year":2024},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:42.274637Z"},"links":{"citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:161ad728b249c8ec9a7f5143e0fcc83a6ca880eaa0f377f0fdc2d420edbba44e","observation_id":"4f3d0a31-25bb-492d-a763-2c477de4a226","resolution":{"observed_at":"2026-08-07T15:34:46.268970Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05530","last_updated":"2024-12-16T17:39:39Z","snapshot_observed_at":"2026-07-06T17:41:42.995949Z","submitted_at":"2024-03-08T18:54:20Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05530","snapshot_observed_at":"2026-08-07T15:34:42.341438Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context.arXiv preprint arXiv:2403.05530, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:42.341438Z"},"links":{"cited_paper":"/paper/2403.05530","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:52d315b0bb3a5a232290e8bd9e9ea3f2374543e682cde4b8d3c8dd1be5f3924b","observation_id":"4cbf0ae1-1f36-4988-99b6-e62437f5ba40","resolution":{"observed_at":"2026-08-07T15:34:42.341438Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-08-07T15:34:42.438272Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:42.438272Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:f64995f222db2a4ae0c492169dd80ba55f9a04bd3e413bb71d4e3655c40e131f","observation_id":"5f383376-799c-40e8-b0ff-10aae3869e41","resolution":{"observed_at":"2026-08-07T15:34:42.438272Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T15:34:42.546538Z","title":"Videochat: Chat-centric video understanding.arXiv preprint arXiv:2305.06355, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:42.546538Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:d8cff89c3defd5bf3cee4aa3794e8ce97f4ec5f9c355a0de656acccbf2846e4d","observation_id":"3f9b5974-c0ef-464f-93fd-7d8253ab8b18","resolution":{"observed_at":"2026-08-07T15:34:42.546538Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05424","last_updated":"2024-06-10T01:36:53Z","snapshot_observed_at":"2026-07-06T15:40:24.127663Z","submitted_at":"2023-06-08T17:59:56Z","title":"Video-ChatGPT: Towards Detailed Video Understanding via Large Vision and Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05424","snapshot_observed_at":"2026-08-07T15:34:42.650236Z","title":"Video-chatgpt: Towards detailed video understanding via large vision and language models.arXiv preprint arXiv:2306.05424, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:42.650236Z"},"links":{"cited_paper":"/paper/2306.05424","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:5b0c6d7fcc23d22f620689960133a96635c747c97ae76e37b0ddf9e35f99584b","observation_id":"3fe5c5d6-7f0e-4d57-a531-0314a6b3bbd8","resolution":{"observed_at":"2026-08-07T15:34:42.650236Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.01483","last_updated":"2024-11-15T06:31:44Z","snapshot_observed_at":"2026-07-06T18:08:56.480769Z","submitted_at":"2024-05-02T17:14:57Z","title":"MANTIS: Interleaved Multi-Image Instruction Tuning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.01483","snapshot_observed_at":"2026-08-07T15:34:42.725167Z","title":"Mantis: Interleaved multi-image instruction tuning.arXiv preprint arXiv:2405.01483, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:42.725167Z"},"links":{"cited_paper":"/paper/2405.01483","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:c247c972ba4eed78faa2bc5b1f9e8bcd39b76c13d8a20a583efde52cf285f49c","observation_id":"f9db9ddf-06ec-471d-8a9e-933d2d1fbcb8","resolution":{"observed_at":"2026-08-07T15:34:42.725167Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.02858","last_updated":"2023-10-25T06:23:31Z","snapshot_observed_at":"2026-07-06T15:38:39.712379Z","submitted_at":"2023-06-05T13:17:27Z","title":"Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.02858","snapshot_observed_at":"2026-08-07T15:34:42.813436Z","title":"Video-llama: An instruction-tuned audio-visual language model for video understanding.arXiv preprint arXiv:2306.02858, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:42.813436Z"},"links":{"cited_paper":"/paper/2306.02858","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:cb6290ded2a322686993acf0cb91f49b71221c18e87e809fa02410336ec0c8af","observation_id":"cc29b80a-afa0-4295-88e0-6ee1ce4abb53","resolution":{"observed_at":"2026-08-07T15:34:42.813436Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:34:42.886209Z","title":"Mvbench: A comprehensive multi-modal video understanding benchmark","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:42.886209Z"},"links":{"citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:0d74369cacb8a754e09ea3daf192ead4c7d9616312662e279c12aac19d48a92f","observation_id":"911ec23a-c43c-4dbf-989f-a33908344c6f","resolution":{"observed_at":"2026-08-07T15:34:42.886209Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14023","last_updated":"2024-08-26T05:27:14Z","snapshot_observed_at":"2026-08-07T05:22:51.503928Z","submitted_at":"2024-08-26T05:27:14Z","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.14023","snapshot_observed_at":"2026-08-07T15:34:42.981605Z","title":"Video-ccam: Enhancing video-language understanding with causal cross-attention masks for short and long videos.arXiv preprint arXiv:2408.14023, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:42.981605Z"},"links":{"cited_paper":"/paper/2408.14023","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:39a6d086de03298ee315b5d5906b6cbe083ff4c17e575bc157bc4ed7cdf478e1","observation_id":"ce17acaf-da91-44b8-a33a-65f9f5f53e0c","resolution":{"observed_at":"2026-08-07T15:34:42.981605Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:34:43.096309Z","title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:43.096309Z"},"links":{"citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:f5ee59a0b6559ddf434868b172d64d108fd1fcb97cbde4e6530ee97aa496e1b7","observation_id":"b0cc6776-42c0-4274-841d-926965e00b77","resolution":{"observed_at":"2026-08-07T15:34:43.096309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.17434","last_updated":"2024-10-22T21:21:37Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-22T21:21:37Z","title":"LongVU: Spatiotemporal Adaptive Compression for Long Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.17434","snapshot_observed_at":"2026-08-07T15:34:43.173823Z","title":"Longvu: Spa- tiotemporal adaptive compression for long video-language understanding.arXiv preprint arXiv:2410.17434, 2024","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:43.173823Z"},"links":{"cited_paper":"/paper/2410.17434","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:1ac6ef651008241601223254f7e1b88cf58b0093f59a3cbc001cdf4bb3562edf","observation_id":"4cac2fae-867a-4ca3-81ec-80b4cd26c169","resolution":{"observed_at":"2026-08-07T15:34:43.173823Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.14485","last_updated":"2024-12-10T12:45:31Z","snapshot_observed_at":"2026-07-06T19:19:28.953455Z","submitted_at":"2024-09-22T15:13:31Z","title":"Video-XL: Extra-Long Vision Language Model for Hour-Scale Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.14485","snapshot_observed_at":"2026-08-07T15:34:43.251039Z","title":"Video-xl: Extra-long vision language model for hour-scale video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:43.251039Z"},"links":{"cited_paper":"/paper/2409.14485","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:3de4e5aa223864533b8d68d2649d4f0b85ea58c079c61f6b54a42e06c6d06c26","observation_id":"5523b6de-9a63-4e83-8221-c14dd59cc6bc","resolution":{"observed_at":"2026-08-07T15:34:43.251039Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.00752","last_updated":"2024-05-31T17:55:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-01T18:01:34Z","title":"Mamba: Linear-Time Sequence Modeling with Selective State Spaces","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.00752","snapshot_observed_at":"2026-08-07T15:34:43.353762Z","title":"Mamba: Linear-time sequence modeling with selective state spaces","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:43.353762Z"},"links":{"cited_paper":"/paper/2312.00752","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:513b7076d05d6b944653b7f801ac2c55f2b070b48e136557da55db3ebe75c91a","observation_id":"a56cf553-118f-4410-91a7-10e529619f69","resolution":{"observed_at":"2026-08-07T15:34:43.353762Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:34:43.440428Z","title":"Vript: A video is worth thousands of words.Advances in Neural Information Processing Systems, 37:57240–57261, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:43.440428Z"},"links":{"citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:cce8e37a957116e5570d56f9f1cc82c09f5e039571bf3ab8f75b610706acd76a","observation_id":"7d083842-2e43-4ed1-97ad-f4013fb07b65","resolution":{"observed_at":"2026-08-07T15:34:43.440428Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.01258","last_updated":"2024-04-02T12:47:49Z","snapshot_observed_at":"2026-08-08T06:22:53.769876Z","submitted_at":"2024-04-01T17:28:16Z","title":"Direct Preference Optimization of Video Large Multimodal Models from Language Model Reward","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.01258","snapshot_observed_at":"2026-08-07T15:34:43.506440Z","title":"Direct preference optimization of video large multimodal models from language model reward.arXiv preprint arXiv:2404.01258, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:43.506440Z"},"links":{"cited_paper":"/paper/2404.01258","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:966671948969f67b18ec7c434404a918037f139fa3ef518f1d2d8f414b91c765","observation_id":"782cb605-d43c-49b8-be75-4e6046766571","resolution":{"observed_at":"2026-08-07T15:34:43.506440Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:34:43.574670Z","title":"Video question answering via gradually refined attention over appearance and motion","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:43.574670Z"},"links":{"citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:8df978574716e14ff12155a0476b3278e9404ebdd3918f02ff5dd1fd538a6634","observation_id":"085e1bbd-1e7e-441c-bca8-6507ba764d3f","resolution":{"observed_at":"2026-08-07T15:34:43.574670Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:34:43.650831Z","title":"Activitynet-qa: A dataset for understanding complex web videos via question answering","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:43.650831Z"},"links":{"citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:3940ac957a92a9a6dfd600ede0536c5ca2bf9c1d5925ec867929eab94cb9e5ab","observation_id":"0d2e9a99-d3cd-4e3a-bbe4-0395d87eacb7","resolution":{"observed_at":"2026-08-07T15:34:43.650831Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00476","last_updated":"2024-06-03T04:13:39Z","snapshot_observed_at":"2026-08-04T21:17:37.211833Z","submitted_at":"2024-03-01T12:02:19Z","title":"TempCompass: Do Video LLMs Really Understand Videos?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00476","snapshot_observed_at":"2026-08-07T15:34:43.723261Z","title":"Tempcompass: Do video llms really understand videos?arXiv preprint arXiv:2403.00476, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:43.723261Z"},"links":{"cited_paper":"/paper/2403.00476","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:a4d1705aff41b5839db2dcde4fc32e145a9edc862474c8f84a325b42820c0885","observation_id":"647e48b4-0e88-4c5d-82ce-b6601f34a514","resolution":{"observed_at":"2026-08-07T15:34:43.723261Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-08-07T15:34:43.788047Z","title":"Videovista: A versatile benchmark for video understanding and reasoning.arXiv preprint arXiv:2406.11303, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:43.788047Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:3cee0a0d7ab97f77aa63d6eb98ae8e03e159fe491573636e28df3d7bd3d0a5ae","observation_id":"29aaa453-09ca-4625-84b3-d9457014c874","resolution":{"observed_at":"2026-08-07T15:34:43.788047Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.04368","last_updated":"2024-11-07T01:58:42Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-11-07T01:58:42Z","title":"Measuring short-form factuality in large language models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.04368","snapshot_observed_at":"2026-08-07T15:34:43.846476Z","title":"Measuring short-form factuality in large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:43.846476Z"},"links":{"cited_paper":"/paper/2411.04368","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:01e7dfa3512dc2e5bb6e429c605a219c3076921b2f4f092220ebed1227b3de86","observation_id":"cee2d8f7-c7d7-4872-8ffd-955bf72fc76f","resolution":{"observed_at":"2026-08-07T15:34:43.846476Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.18923","last_updated":"2025-08-13T17:44:18Z","snapshot_observed_at":"2026-08-07T16:40:29.716914Z","submitted_at":"2025-03-24T17:46:09Z","title":"Video SimpleQA: Towards Factuality Evaluation in Large Video Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.18923","snapshot_observed_at":"2026-08-07T15:34:43.911417Z","title":"Video simpleqa: Towards factuality evaluation in large video language models.arXiv preprint arXiv:2503.18923, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:43.911417Z"},"links":{"cited_paper":"/paper/2503.18923","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:c54a6b639e53eea603d729916d574e78bf20b11e098cfd7efd28343639489cf5","observation_id":"6ab987bd-e7c2-4885-ba42-fd30ab5413d9","resolution":{"observed_at":"2026-08-07T15:34:43.911417Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15594","last_updated":"2025-10-19T10:32:43Z","snapshot_observed_at":"2026-08-02T10:23:50.881300Z","submitted_at":"2024-11-23T16:03:35Z","title":"A Survey on LLM-as-a-Judge","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15594","snapshot_observed_at":"2026-08-07T15:34:44.002274Z","title":"A survey on llm-as-a-judge.arXiv preprint arXiv:2411.15594, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:44.002274Z"},"links":{"cited_paper":"/paper/2411.15594","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:d58c029e3e7b1a05280cb94e6451544f39feaba4f68b496bf669eeccefda16e9","observation_id":"e73cd584-1302-45ee-9b4f-ad87622b26ea","resolution":{"observed_at":"2026-08-07T15:34:44.002274Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:34:44.072681Z","title":"Judging llm-as-a-judge with mt-bench and chatbot arena.Advances in Neural Information Processing Systems, 36:46595–46623, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:44.072681Z"},"links":{"citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:c2b445eb1a73906711c7945f492594d954cb521570575242589a5ae5e7c48947","observation_id":"490bec7a-661d-4ac6-85fa-759897b69dbd","resolution":{"observed_at":"2026-08-07T15:34:44.072681Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:34:45.940665Z","title":"Gpt-4o.https://openai.com/index/hello-gpt-4o/, 2024","venue":null,"work_id":"a61df4a2-19ea-4907-b04f-58a2f3cbfcef","year":2024},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:44.129970Z"},"links":{"citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:7ce3f7770dde9cfcd0ec9ca206e73523cea5bb85f1ce38d9eed345c0d845a2b5","observation_id":"c76730be-8774-46a2-a7c2-37b02a43f9a7","resolution":{"observed_at":"2026-08-07T15:34:46.047734Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:34:45.826770Z","title":"Gpt-4o mini: Advancing cost-efficient intelligence, July 2024","venue":null,"work_id":"e0f5ebae-2c82-444d-9155-fcb90a7037c1","year":2024},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:44.192407Z"},"links":{"citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:d2d033081ec02454ad1526d8ed6dbc352828439342f84f55b0e5f27d23dc2942","observation_id":"4d87df07-319c-4926-b206-0ca0809dfd24","resolution":{"observed_at":"2026-08-07T15:34:45.877992Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:34:45.783424Z","title":"Introducing gpt-4.1 in the api, April 2025","venue":null,"work_id":"c6f83731-3023-4cba-84af-7d9fc7f29875","year":2025},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:44.250379Z"},"links":{"citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:867fedd0783366e60c2be8b587c656a28dce8205982a632689acaa3e11a553dd","observation_id":"d0f93d1e-fae4-425a-bff1-0880eecedd1f","resolution":{"observed_at":"2026-08-07T15:34:45.790494Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:34:45.699211Z","title":"Gemini 2.5: Our most intelligent ai model, March","venue":null,"work_id":"34571e59-8efa-47ee-aa4f-f50cc2c1b43c","year":null},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:44.319389Z"},"links":{"citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:486b2f674dbfab3d46a46062e0c30a4af837f32b57b7462cb485d561b76c22e6","observation_id":"f669b362-f506-4ab3-97ad-60abaef86525","resolution":{"observed_at":"2026-08-07T15:34:45.725060Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12191","last_updated":"2024-10-03T15:54:49Z","snapshot_observed_at":"2026-08-06T05:35:29.109022Z","submitted_at":"2024-09-18T17:59:32Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12191","snapshot_observed_at":"2026-08-07T15:34:44.468889Z","title":"Qwen2-vl: Enhancing vision-language model’s perception of the world at any resolution.arXiv preprint arXiv:2409.12191, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:44.468889Z"},"links":{"cited_paper":"/paper/2409.12191","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:5ee9fa91ac3bc5b126f775f2594fedcaaace3170c3cce214e801ed01cdb65367","observation_id":"1bebcfb6-ddfd-44f2-a173-13b10dc86f4f","resolution":{"observed_at":"2026-08-07T15:34:44.468889Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.05271","last_updated":"2025-09-26T12:52:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-06T18:57:08Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.05271","snapshot_observed_at":"2026-08-07T15:34:44.518133Z","title":"Expanding performance boundaries of open-source multimodal models with model, data, and test-time scaling.arXiv preprint arXiv:2412.05271, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:44.518133Z"},"links":{"cited_paper":"/paper/2412.05271","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:5b8a79c57737ebdb9b8de982bcdcfb8d54083502e4c198c0c2f63d5e7692c6a9","observation_id":"5d0872ad-c2e1-40e5-a047-d42e50b38f36","resolution":{"observed_at":"2026-08-07T15:34:44.518133Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.10479","last_updated":"2025-04-19T03:47:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-14T17:59:25Z","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.10479","snapshot_observed_at":"2026-08-07T15:34:44.591263Z","title":"Internvl3: Exploring advanced training and test-time recipes for open-source multimodal models.arXiv preprint arXiv:2504.10479, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:44.591263Z"},"links":{"cited_paper":"/paper/2504.10479","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:79b7b75243fd4b1a57f76763b93fec633786e6f1921401e4ec3d9f5c2f08fca0","observation_id":"e2f50624-c532-4090-9107-576c2a735bad","resolution":{"observed_at":"2026-08-07T15:34:44.591263Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.08905","last_updated":"2024-12-12T03:37:41Z","snapshot_observed_at":"2026-08-05T04:04:21.846023Z","submitted_at":"2024-12-12T03:37:41Z","title":"Phi-4 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.08905","snapshot_observed_at":"2026-08-07T15:34:44.670107Z","title":"Phi-4 technical report.arXiv preprint arXiv:2412.08905, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:44.670107Z"},"links":{"cited_paper":"/paper/2412.08905","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:bb34b6bbac85b4ff5106b065a64bd94e793f5e20f7d7a34c82126099ac517872","observation_id":"fb22c9ce-78de-42ef-8966-21a2949b6f12","resolution":{"observed_at":"2026-08-07T15:34:44.670107Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:34:44.746446Z","title":"Longllava: Scaling multi-modal llms to 1000 images efficiently via a hybrid architecture.arXiv preprint arXiv:2409.02889, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:44.746446Z"},"links":{"citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:7f7576641843b9cdfbb34362090ece1129582d090fe5350a10a03ef1d186714c","observation_id":"25113abc-91cd-44f7-93f6-2d069f44df5a","resolution":{"observed_at":"2026-08-07T15:34:44.746446Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.08691","last_updated":"2023-07-17T17:50:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-17T17:50:36Z","title":"FlashAttention-2: Faster Attention with Better Parallelism and Work Partitioning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.08691","snapshot_observed_at":"2026-08-07T15:34:44.818600Z","title":"Flashattention-2: Faster attention with better parallelism and work partitioning.arXiv preprint arXiv:2307.08691, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:44.818600Z"},"links":{"cited_paper":"/paper/2307.08691","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:4b59f41f46c5217623d99dc4a2e0780cdba533db75e85ffc42bd4f6404f1d991","observation_id":"f16b9d63-0c2a-4bd7-8b01-f283f01c91d8","resolution":{"observed_at":"2026-08-07T15:34:44.818600Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:34:45.535225Z","title":"Keep\" if the question can be answered by someone who has watched the video, even if the answer requires reasoning or summarizing visual or auditory evidence. -","venue":null,"work_id":"40c60ec7-823e-4c45-b961-70114fd9cda4","year":1998},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:44.884301Z"},"links":{"citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:581c4fcc98ab4002b44478dabe09fc7a9563cdddc51b6df2553859470a24de19","observation_id":"ff54a88f-14c1-4d97-958e-95fd6b6afaba","resolution":{"observed_at":"2026-08-07T15:34:45.565691Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:34:45.619897Z","title":"Accessed: 2025-05-08","venue":null,"work_id":"bb74e6a2-2538-4b57-b1d0-fc258e4848d5","year":2025},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:44.396406Z"},"links":{"citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:26ec5fcc99cb1035e3f874d72591a4d818d1ef36e1fa139f0389d53801ed86c5","observation_id":"0a5fdf07-81df-48d0-b405-ba843d7a7362","resolution":{"observed_at":"2026-08-07T15:34:45.668732Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation"},"reference_resolution":{"displayed":55,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":48,"verified_exact":0,"verified_fuzzy":7},"total_outbound_references":55},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 55 of 55 outbound references and 10 inbound Pith citation observations for arXiv:2505.14640."}