{"as_of":"2026-08-09T09:59:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:ae349dc5996b9abe8f97928861e5987ab603484732a93c0bd537381e050f1625","coverage":[{"denominator":104,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T04:45:10.764292Z","state":"measured"},{"denominator":118,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":118,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":18,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":18,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T22:10:27.260004Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T12:49:53.095154Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09943","snapshot_observed_at":"2026-08-06T22:10:27.260004Z","title":"Causalvqa: A physically grounded causal reasoning benchmark for video models.arXiv preprint arXiv:2506.09943,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.22355","last_updated":"2025-07-07T15:42:11Z","snapshot_observed_at":"2026-08-07T04:25:28.195802Z","submitted_at":"2025-06-27T16:05:34Z","title":"Embodied AI Agents: Modeling the World","version":3},"reference_index":2016,"source":"pdf_text","source_observed_at":"2026-08-06T22:10:27.260004Z"},"links":{"cited_paper":"/paper/2506.09943","citing_paper":"/paper/2506.22355"},"observation_digest":"sha256:34ec8d89d25ac55fc5ac9874a2bd2ea54193aedf18102a74df5d6b3c847bc127","observation_id":"06d92058-3d59-46b4-8231-8bdbdf7ab100","resolution":{"observed_at":"2026-08-06T22:10:27.260004Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09943","snapshot_observed_at":"2026-08-05T14:41:42.914172Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.21010","last_updated":"2026-07-04T17:47:47Z","snapshot_observed_at":"2026-08-09T01:40:51.113440Z","submitted_at":"2025-08-28T17:10:53Z","title":"ChainReaction: Causal Chain-Guided Reasoning for Modular and Explainable Causal-Why Video Question Answering","version":3},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-05T14:41:42.914172Z"},"links":{"cited_paper":"/paper/2506.09943","citing_paper":"/paper/2508.21010"},"observation_digest":"sha256:e68c038de4da584be45911e0ba21e051bdb42301e54e62c8e07625912c7ae3ba","observation_id":"c171c473-1f93-49b4-81ef-dec3e8bd2ca3","resolution":{"observed_at":"2026-08-05T14:41:42.914172Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09943","snapshot_observed_at":"2026-08-03T21:14:09.234652Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2511.17649","last_updated":"2026-07-07T04:04:31Z","snapshot_observed_at":"2026-08-07T15:20:09.839961Z","submitted_at":"2025-11-20T09:52:20Z","title":"SWITCH: Benchmarking Modeling and Handling of Tangible Interfaces in Long-horizon Embodied Scenarios","version":5},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-03T21:14:09.234652Z"},"links":{"cited_paper":"/paper/2506.09943","citing_paper":"/paper/2511.17649"},"observation_digest":"sha256:04fe286274622c05ce5d76f74bd4d5cc1fda2c40d2301bd5821e7df018dfb89c","observation_id":"708bd6da-b77d-409a-94e2-a59e5fbe1210","resolution":{"observed_at":"2026-08-03T21:14:09.234652Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"cited_work":{"arxiv_id":"2506.09943","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.09943","snapshot_observed_at":"2026-07-04T12:49:53.095154Z","title":"Causalvqa: A physically grounded causal reasoning benchmark for video models","venue":null,"work_id":"4545b7fc-2bcc-44f3-a395-c6684aede4c7","year":2025},"citing_paper":{"arxiv_id":"2603.03944","last_updated":"2026-04-03T20:11:12Z","snapshot_observed_at":"2026-07-06T22:47:46.409064Z","submitted_at":"2026-03-04T11:09:39Z","title":"SCP: Spatial Causal Prediction in Video","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-15T16:47:44.523606Z"},"links":{"cited_paper":"/paper/2506.09943","citing_paper":"/paper/2603.03944"},"observation_digest":"sha256:3e76c37ca6f5536f4bceb5d0f562702cc16b5136e05a9f16a5a70859e50ef75a","observation_id":"3496926f-5e34-48c2-8842-8cf1f30fa475","resolution":{"observed_at":"2026-05-15T16:50:11.238432Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"cited_work":{"arxiv_id":"2506.09943","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.09943","snapshot_observed_at":"2026-07-04T12:49:53.095154Z","title":"Causalvqa: A physically grounded causal reasoning benchmark for video models","venue":null,"work_id":"4545b7fc-2bcc-44f3-a395-c6684aede4c7","year":2025},"citing_paper":{"arxiv_id":"2604.04707","last_updated":"2026-05-25T06:28:44Z","snapshot_observed_at":"2026-07-13T09:42:22.607962Z","submitted_at":"2026-04-06T14:19:48Z","title":"OpenWorldLib: A Unified Codebase and Definition of Advanced World Models","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-10T19:36:42.100191Z"},"links":{"cited_paper":"/paper/2506.09943","citing_paper":"/paper/2604.04707"},"observation_digest":"sha256:200642a0f0b523038be63816286902614505dfbd6945e1870b93d2d8d8d9eb9c","observation_id":"f67e0dff-1b29-4b5a-b2f2-c8b4d3cd176e","resolution":{"observed_at":"2026-05-10T22:45:49.318294Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09943","snapshot_observed_at":"2026-07-13T09:42:23.808691Z","title":"Causalvqa: A physically grounded causal reasoning benchmark for video models.arXiv preprint arXiv:2506.09943, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2604.04707","last_updated":"2026-05-25T06:28:44Z","snapshot_observed_at":"2026-07-13T09:42:22.607962Z","submitted_at":"2026-04-06T14:19:48Z","title":"OpenWorldLib: A Unified Codebase and Definition of Advanced World Models","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-07-13T09:42:23.808691Z"},"links":{"cited_paper":"/paper/2506.09943","citing_paper":"/paper/2604.04707"},"observation_digest":"sha256:c535a6822e7c157d7987af6dfc402efce35d7d4cf5c98353c7c9f458ac92dc62","observation_id":"b8071a87-fb3f-4568-92b6-9a6028d83d0b","resolution":{"observed_at":"2026-07-13T09:42:23.808691Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"cited_work":{"arxiv_id":"2506.09943","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.09943","snapshot_observed_at":"2026-07-04T12:49:53.095154Z","title":"Causalvqa: A physically grounded causal reasoning benchmark for video models","venue":null,"work_id":"4545b7fc-2bcc-44f3-a395-c6684aede4c7","year":2025},"citing_paper":{"arxiv_id":"2605.01657","last_updated":"2026-05-03T00:52:51Z","snapshot_observed_at":"2026-08-08T14:11:45.861701Z","submitted_at":"2026-05-03T00:52:51Z","title":"Act2See: Emergent Active Visual Perception for Video Reasoning","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-08T19:34:53.683729Z"},"links":{"cited_paper":"/paper/2506.09943","citing_paper":"/paper/2605.01657"},"observation_digest":"sha256:60b9789439c9a5956cc1ec2e174823a87e9e0e0572b783d2d9cb6ff29c0226fb","observation_id":"d2703f24-39ad-40c5-94ad-d342a1f99f3c","resolution":{"observed_at":"2026-05-09T05:45:22.912863Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"cited_work":{"arxiv_id":"2506.09943","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.09943","snapshot_observed_at":"2026-07-04T12:49:53.095154Z","title":"Causalvqa: A physically grounded causal reasoning benchmark for video models","venue":null,"work_id":"4545b7fc-2bcc-44f3-a395-c6684aede4c7","year":2025},"citing_paper":{"arxiv_id":"2605.23216","last_updated":"2026-07-02T06:54:56Z","snapshot_observed_at":"2026-08-03T00:30:19.785022Z","submitted_at":"2026-05-22T04:19:29Z","title":"CaST-Bench: Benchmarking Causal Chain-Grounded Spatio-Temporal Reasoning for Video Question Answering","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-25T04:54:23.077914Z"},"links":{"cited_paper":"/paper/2506.09943","citing_paper":"/paper/2605.23216"},"observation_digest":"sha256:a26f43e4fe291ff4950a28c03c2aceb16d25cf53225358f6aff9c16ae17ce984","observation_id":"e1303ebf-f7a3-4408-9466-2d56c2c0114f","resolution":{"observed_at":"2026-05-25T04:55:23.100745Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"cited_work":{"arxiv_id":"2506.09943","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.09943","snapshot_observed_at":"2026-07-04T12:49:53.095154Z","title":"Causalvqa: A physically grounded causal reasoning benchmark for video models","venue":null,"work_id":"4545b7fc-2bcc-44f3-a395-c6684aede4c7","year":2025},"citing_paper":{"arxiv_id":"2605.23216","last_updated":"2026-07-02T06:54:56Z","snapshot_observed_at":"2026-08-03T00:30:19.785022Z","submitted_at":"2026-05-22T04:19:29Z","title":"CaST-Bench: Benchmarking Causal Chain-Grounded Spatio-Temporal Reasoning for Video Question Answering","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-07-04T00:41:02.284215Z"},"links":{"cited_paper":"/paper/2506.09943","citing_paper":"/paper/2605.23216"},"observation_digest":"sha256:944140a7b0cac44ea1668efa0ab30fe74ff28dc4d970d8e7f7326293fcac2ba0","observation_id":"1cdced80-7d96-455c-b023-13f2e7a1eb72","resolution":{"observed_at":"2026-07-04T00:49:17.413207Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"cited_work":{"arxiv_id":"2506.09943","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.09943","snapshot_observed_at":"2026-07-04T12:49:53.095154Z","title":"Causalvqa: A physically grounded causal reasoning benchmark for video models","venue":null,"work_id":"4545b7fc-2bcc-44f3-a395-c6684aede4c7","year":2025},"citing_paper":{"arxiv_id":"2605.23699","last_updated":"2026-05-22T14:51:22Z","snapshot_observed_at":"2026-07-06T23:33:53.870540Z","submitted_at":"2026-05-22T14:51:22Z","title":"CRONOS: Benchmarking Counterfactual Physical Consistency in Video Models","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-25T04:39:22.400458Z"},"links":{"cited_paper":"/paper/2506.09943","citing_paper":"/paper/2605.23699"},"observation_digest":"sha256:09897ecee804cf4bf5e76c18220686330331e8d130f86d56ee7dec2feeb94b20","observation_id":"9b9620ea-20ef-4c4e-a55d-1c0ef7d82bd4","resolution":{"observed_at":"2026-05-25T04:40:23.275017Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"cited_work":{"arxiv_id":"2506.09943","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.09943","snapshot_observed_at":"2026-07-04T12:49:53.095154Z","title":"Causalvqa: A physically grounded causal reasoning benchmark for video models","venue":null,"work_id":"4545b7fc-2bcc-44f3-a395-c6684aede4c7","year":2025},"citing_paper":{"arxiv_id":"2605.27589","last_updated":"2026-05-26T19:02:26Z","snapshot_observed_at":"2026-08-05T09:42:53.562841Z","submitted_at":"2026-05-26T19:02:26Z","title":"What-If World: A Causal Benchmark for General World Models in Embodied Scenarios","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-29T18:23:22.987086Z"},"links":{"cited_paper":"/paper/2506.09943","citing_paper":"/paper/2605.27589"},"observation_digest":"sha256:57c6a4e648029d83a871d6bd9f1ba53af4b89447fa1988edcf053ce706396fff","observation_id":"5447d01b-d4c6-4827-988c-d999c60e73cf","resolution":{"observed_at":"2026-06-29T18:23:50.387373Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"cited_work":{"arxiv_id":"2506.09943","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.09943","snapshot_observed_at":"2026-07-04T12:49:53.095154Z","title":"Causalvqa: A physically grounded causal reasoning benchmark for video models","venue":null,"work_id":"4545b7fc-2bcc-44f3-a395-c6684aede4c7","year":2025},"citing_paper":{"arxiv_id":"2605.30346","last_updated":"2026-05-28T17:59:51Z","snapshot_observed_at":"2026-08-04T10:14:05.552405Z","submitted_at":"2026-05-28T17:59:51Z","title":"YoCausal: How Far is Video Generation from World Model? A Causality Perspective","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-06-29T08:27:03.674229Z"},"links":{"cited_paper":"/paper/2506.09943","citing_paper":"/paper/2605.30346"},"observation_digest":"sha256:dbcce8ac8f0e20a08c4179a57fa4dfb5b8a272bc5ba4e4420993a4024c97cc7f","observation_id":"db74685d-7bbd-4d2a-adb2-3fea251bfeaa","resolution":{"observed_at":"2026-06-29T08:33:15.575730Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"cited_work":{"arxiv_id":"2506.09943","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.09943","snapshot_observed_at":"2026-07-04T12:49:53.095154Z","title":"Causalvqa: A physically grounded causal reasoning benchmark for video models","venue":null,"work_id":"4545b7fc-2bcc-44f3-a395-c6684aede4c7","year":2025},"citing_paper":{"arxiv_id":"2606.02800","last_updated":"2026-06-23T17:33:32Z","snapshot_observed_at":"2026-07-06T23:43:07.940839Z","submitted_at":"2026-06-01T19:12:30Z","title":"Cosmos 3: Omnimodal World Models for Physical AI","version":4},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-28T15:08:33.957835Z"},"links":{"cited_paper":"/paper/2506.09943","citing_paper":"/paper/2606.02800"},"observation_digest":"sha256:5402c6519ecfc02dc19d6b5e2526ad6c3924b18580df03c3182c6e72c7aa0d08","observation_id":"a1e8a19a-7040-462b-b9c1-5a55972fb968","resolution":{"observed_at":"2026-07-01T22:36:18.020144Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"cited_work":{"arxiv_id":"2506.09943","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.09943","snapshot_observed_at":"2026-07-04T12:49:53.095154Z","title":"Causalvqa: A physically grounded causal reasoning benchmark for video models","venue":null,"work_id":"4545b7fc-2bcc-44f3-a395-c6684aede4c7","year":2025},"citing_paper":{"arxiv_id":"2606.05966","last_updated":"2026-06-04T10:07:05Z","snapshot_observed_at":"2026-07-06T23:45:51.910263Z","submitted_at":"2026-06-04T10:07:05Z","title":"Causal Scaffolding for Physical Reasoning: A Benchmark for Causally-Informed Physical World Understanding in VLMs","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-27T23:15:38.962013Z"},"links":{"cited_paper":"/paper/2506.09943","citing_paper":"/paper/2606.05966"},"observation_digest":"sha256:c7e3a58b72d4a2d2fe1227be497394efd77a1f38d9d9c7d0c5113669c3d9c0d4","observation_id":"af680def-7987-4cee-a9ab-6fdf05ead8c0","resolution":{"observed_at":"2026-07-02T15:57:06.739294Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"cited_work":{"arxiv_id":"2506.09943","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.09943","snapshot_observed_at":"2026-07-04T12:49:53.095154Z","title":"Causalvqa: A physically grounded causal reasoning benchmark for video models","venue":null,"work_id":"4545b7fc-2bcc-44f3-a395-c6684aede4c7","year":2025},"citing_paper":{"arxiv_id":"2606.15032","last_updated":"2026-06-28T03:48:42Z","snapshot_observed_at":"2026-07-06T23:52:28.557859Z","submitted_at":"2026-06-13T00:21:21Z","title":"How Should World Models Be Evaluated for Embodied Decision-Making? A Decision-Making-Centric Position","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-30T10:31:37.108792Z"},"links":{"cited_paper":"/paper/2506.09943","citing_paper":"/paper/2606.15032"},"observation_digest":"sha256:2c68d046c23b630891e9d139d735224867c754958b1f6be6e4cedb1179d080c8","observation_id":"8b225745-5f30-4d67-8a2c-4d446751cb5f","resolution":{"observed_at":"2026-06-30T10:34:36.404842Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"cited_work":{"arxiv_id":"2506.09943","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.09943","snapshot_observed_at":"2026-07-04T12:49:53.095154Z","title":"Causalvqa: A physically grounded causal reasoning benchmark for video models","venue":null,"work_id":"4545b7fc-2bcc-44f3-a395-c6684aede4c7","year":2025},"citing_paper":{"arxiv_id":"2606.26694","last_updated":"2026-06-28T04:16:09Z","snapshot_observed_at":"2026-07-07T00:01:01.437499Z","submitted_at":"2026-06-25T07:27:09Z","title":"PhysEditWorld: A Large-Scale Dataset Toward Physics-Editable World Models","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-06-26T05:46:21.198781Z"},"links":{"cited_paper":"/paper/2506.09943","citing_paper":"/paper/2606.26694"},"observation_digest":"sha256:a10be37ff1a326fbf16966e329e23a4fc37203cc7678a4e36e206113b13f639a","observation_id":"dcb352d9-db36-4f16-9b37-98c599dc3ae0","resolution":{"observed_at":"2026-07-04T12:49:53.096775Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"cited_work":{"arxiv_id":"2506.09943","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.09943","snapshot_observed_at":"2026-07-04T12:49:53.095154Z","title":"Causalvqa: A physically grounded causal reasoning benchmark for video models","venue":null,"work_id":"4545b7fc-2bcc-44f3-a395-c6684aede4c7","year":2025},"citing_paper":{"arxiv_id":"2606.26694","last_updated":"2026-06-28T04:16:09Z","snapshot_observed_at":"2026-07-07T00:01:01.437499Z","submitted_at":"2026-06-25T07:27:09Z","title":"PhysEditWorld: A Large-Scale Dataset Toward Physics-Editable World Models","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-06-30T10:19:06.268547Z"},"links":{"cited_paper":"/paper/2506.09943","citing_paper":"/paper/2606.26694"},"observation_digest":"sha256:d6645e8d02c7d7a997e514b6739c27d2f6e47fe42220a7581a2171bf27e29dde","observation_id":"1b8029b4-2516-49b7-a44f-e357bfb86ed4","resolution":{"observed_at":"2026-06-30T12:04:39.227823Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09943","snapshot_observed_at":"2026-07-14T07:27:08.281893Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.11044","last_updated":"2026-07-13T03:13:39Z","snapshot_observed_at":"2026-08-07T01:14:39.218300Z","submitted_at":"2026-07-13T03:13:39Z","title":"RetroHolmes: When Semantic Plausibility Fails Retrospective Physical Process Reasoning","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-07-14T07:27:08.281893Z"},"links":{"cited_paper":"/paper/2506.09943","citing_paper":"/paper/2607.11044"},"observation_digest":"sha256:253ee650e2c5e610bb003ff085570eb9bf91c3d9a1a3f980a2d7801dc473ab31","observation_id":"02a00a76-2403-43e0-a327-2d3c184a685b","resolution":{"observed_at":"2026-07-14T07:27:08.281893Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2506.09943/citation-record","integrity":"/paper/2506.09943/integrity","json":"/paper/2506.09943/citation-record.json","paper":"/paper/2506.09943"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:44:58.464903Z","title":"Social-iq: A question answering benchmark for artificial social intelligence","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T04:44:58.464903Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:4b7fefd6bdfba778e053ad6c0434cd2b872fcbe085c6c7ed27bcf45c82ba8168","observation_id":"e0cdc4cc-f70e-4527-a4b4-291d7cbaf374","resolution":{"observed_at":"2026-08-07T04:44:58.464903Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:44:58.566590Z","title":"Sheng, Wei Emma Zhang, Munazza Zaib, and Ahoud Alhazmi","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T04:44:58.566590Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:49f5ecdc0a39a3177b918351da0a4d718e142f87debc13f70a0bbe6030a88669","observation_id":"3316cd5e-a9bc-468d-a283-05721b3ada39","resolution":{"observed_at":"2026-08-07T04:44:58.566590Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-08-07T04:44:58.634967Z","title":"Qwen2.5-vl technical report, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T04:44:58.634967Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:84213017696db19607365d5677acbad1ef96daeccb3938b7faad2db537c27e93","observation_id":"e6b21193-cbd9-4b2c-bfda-3a8063deceea","resolution":{"observed_at":"2026-08-07T04:44:58.634967Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.05656","last_updated":"2019-08-15T17:58:32Z","snapshot_observed_at":"2026-08-08T13:30:06.897676Z","submitted_at":"2019-08-15T17:58:32Z","title":"PHYRE: A New Benchmark for Physical Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.05656","snapshot_observed_at":"2026-08-07T04:44:58.700144Z","title":"Phyre: A new benchmark for physical reasoning, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T04:44:58.700144Z"},"links":{"cited_paper":"/paper/1908.05656","citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:420e674bbafe1c6167c05080d533b6851438a9879bff8887f13776266987c7cb","observation_id":"9215161d-fcd7-4489-abd8-f504ee9e3adb","resolution":{"observed_at":"2026-08-07T04:44:58.700144Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.03520","last_updated":"2024-10-03T17:24:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-05T17:53:55Z","title":"VideoPhy: Evaluating Physical Commonsense for Video Generation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.03520","snapshot_observed_at":"2026-08-07T04:44:58.786570Z","title":"Videophy: Evaluating physical commonsense for video generation, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T04:44:58.786570Z"},"links":{"cited_paper":"/paper/2406.03520","citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:351d0c876f2826db212c52c8725a87748204d432011851df7f0b659cd6c492ca","observation_id":"0c5f029b-b8bb-4297-9916-6034a2be7c4f","resolution":{"observed_at":"2026-08-07T04:44:58.786570Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.04998","last_updated":"2024-11-07T18:59:16Z","snapshot_observed_at":"2026-07-06T19:46:54.707852Z","submitted_at":"2024-11-07T18:59:16Z","title":"HourVideo: 1-Hour Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.04998","snapshot_observed_at":"2026-08-07T04:44:58.871782Z","title":"Hadzic, Taran Kota, Jimming He, Cristóbal Eyzaguirre, Zane Durante, Manling Li, Jiajun Wu, and Li Fei-Fei","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T04:44:58.871782Z"},"links":{"cited_paper":"/paper/2411.04998","citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:92ec423fda2cf2662deb5a5f56a999a4c8162c8294fe581be52ae6a86d405ab9","observation_id":"ed00242b-b453-4eac-8022-3632c09b33c5","resolution":{"observed_at":"2026-08-07T04:44:58.871782Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.01089","last_updated":"2022-05-02T17:59:13Z","snapshot_observed_at":"2026-08-02T08:10:51.211813Z","submitted_at":"2022-05-02T17:59:13Z","title":"ComPhy: Compositional Physical Reasoning of Objects and Events from Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.01089","snapshot_observed_at":"2026-08-07T04:44:58.928697Z","title":"Tenenbaum, and Chuang Gan","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T04:44:58.928697Z"},"links":{"cited_paper":"/paper/2205.01089","citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:003f1f9882a083b9721e6cb4594cbda5f2a1090de2672cf535cb0ef7d9dfd04c","observation_id":"9cf37a6e-7a0e-41ea-be32-edb88b56de13","resolution":{"observed_at":"2026-08-07T04:44:58.928697Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.13180","last_updated":"2025-07-23T19:22:35Z","snapshot_observed_at":"2026-08-07T23:47:53.537198Z","submitted_at":"2025-04-17T17:59:56Z","title":"PerceptionLM: Open-Access Data and Models for Detailed Visual Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.13180","snapshot_observed_at":"2026-08-07T04:44:59.009056Z","title":"Perceptionlm: Open-access data and models for detailed visual understanding","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T04:44:59.009056Z"},"links":{"cited_paper":"/paper/2504.13180","citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:f1ff9e4847b6f189d2be8b89e3805db0f3f72715afaa592c510e7dd2900dac49","observation_id":"ab285409-aff3-450c-b2d9-708aa346803f","resolution":{"observed_at":"2026-08-07T04:44:59.009056Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.05271","last_updated":"2025-09-26T12:52:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-06T18:57:08Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.05271","snapshot_observed_at":"2026-08-07T04:44:59.108243Z","title":"Expanding performance boundaries of open-source multimodal models with model, data, and test-time scaling, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T04:44:59.108243Z"},"links":{"cited_paper":"/paper/2412.05271","citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:981471d16f8334d984375d2633da84e3cd6a9e186804813ba346c4bb70f47b88","observation_id":"412b0bde-1e4f-4451-95e3-619ecd1959bc","resolution":{"observed_at":"2026-08-07T04:44:59.108243Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1803.09010","last_updated":"2021-12-01T20:29:42Z","snapshot_observed_at":"2026-08-06T13:05:51.883274Z","submitted_at":"2018-03-23T23:22:18Z","title":"Datasheets for Datasets","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1803.09010","snapshot_observed_at":"2026-08-07T04:44:59.171839Z","title":"Datasheets for datasets, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T04:44:59.171839Z"},"links":{"cited_paper":"/paper/1803.09010","citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:f6217335c17d5c18d101bd7fe49069688e98226a8fae7df2dd2c864d738bb8b8","observation_id":"a8615da6-b413-4318-b150-91431ccd7efb","resolution":{"observed_at":"2026-08-07T04:44:59.171839Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:44:59.219836Z","title":"Causal reasoning through intervention.Causal learning: Psychology, philosophy, and computation, 5, 2007","venue":null,"work_id":null,"year":2007},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T04:44:59.219836Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:05f9c85e8428218de072a75d949d44028dd93c517e72158c842fead67b33cbd5","observation_id":"132c9f29-7c22-4bd4-9f7e-7396acf78284","resolution":{"observed_at":"2026-08-07T04:44:59.219836Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.10300","last_updated":"2025-02-05T09:57:59Z","snapshot_observed_at":"2026-07-06T17:03:56.263404Z","submitted_at":"2023-12-16T03:17:30Z","title":"Shot2Story: A New Benchmark for Comprehensive Understanding of Multi-shot Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.10300","snapshot_observed_at":"2026-08-07T04:44:59.302001Z","title":"Shot2story: A new benchmark for comprehensive understanding of multi-shot videos, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T04:44:59.302001Z"},"links":{"cited_paper":"/paper/2312.10300","citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:7652dd22f3a7da5c7f44d37a25a74097f7d003fdf4e054d69415f9082abf7b21","observation_id":"3959a487-d5e7-4256-979c-3260fefde00a","resolution":{"observed_at":"2026-08-07T04:44:59.302001Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:44:59.403213Z","title":"Tgif-qa: Toward spatio-temporal reasoning in visual question answering","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T04:44:59.403213Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:68f74c170c2940cceee7cce33650303356b5e013bf35f8c8fefe6688f9ba1489","observation_id":"2e819dca-8a79-4f5a-acf5-0d98a13143f7","resolution":{"observed_at":"2026-08-07T04:44:59.403213Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1705.06950","last_updated":"2017-05-19T12:07:01Z","snapshot_observed_at":"2026-08-08T17:46:50.107463Z","submitted_at":"2017-05-19T12:07:01Z","title":"The Kinetics Human Action Video Dataset","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1705.06950","snapshot_observed_at":"2026-08-07T04:44:59.485305Z","title":"The kinetics human action video dataset, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T04:44:59.485305Z"},"links":{"cited_paper":"/paper/1705.06950","citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:bab8f95016f7ace6af49510dcff29d3d6579933798c1ce3188628079b8ae30ca","observation_id":"3facfe9c-da71-4bad-93dd-ad7659a6695f","resolution":{"observed_at":"2026-08-07T04:44:59.485305Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:44:59.568462Z","title":"Processing counterfactual and hypothetical conditionals: An fmri investigation.NeuroImage, 72:265–271, 2013","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T04:44:59.568462Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:ca6bdb81848419c305e8aff182ebc82c27a493aa7d91f39117dfcbbc077cc4e8","observation_id":"a81c8cb8-e245-4d8b-9cad-c60f0d609b6b","resolution":{"observed_at":"2026-08-07T04:44:59.568462Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:44:59.648649Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T04:44:59.648649Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:acb0117cd31342a4bbff285eedbddc0bf794b4410abdc6d69ac67fa9af1ebcca","observation_id":"edaaddba-6db6-418d-992d-c6d6dd44a7a7","resolution":{"observed_at":"2026-08-07T04:44:59.648649Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:44:59.817601Z","title":"Lmms-eval: Accelerating the development of large multimoal models, March 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T04:44:59.817601Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:e9fa17474ef5954415886d0ebcfb09d90f107f2955c2751e5343b3fa624192f0","observation_id":"5085d8a3-d028-4c8a-bce9-e5418ec022a7","resolution":{"observed_at":"2026-08-07T04:44:59.817601Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03326","last_updated":"2024-10-26T16:35:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:59:44Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03326","snapshot_observed_at":"2026-08-07T04:44:59.912246Z","title":"Llava-onevision: Easy visual task transfer, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T04:44:59.912246Z"},"links":{"cited_paper":"/paper/2408.03326","citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:95300366106d11de1d990137a39dce7fc2b36e313d9b303f5440af9ede90bead","observation_id":"4a3ded07-18b9-4d13-9b43-8ebbacdf7b01","resolution":{"observed_at":"2026-08-07T04:44:59.912246Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.08632","last_updated":"2024-09-06T11:20:13Z","snapshot_observed_at":"2026-07-06T19:01:27.827878Z","submitted_at":"2024-08-16T09:52:02Z","title":"A Survey on Benchmarks of Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.08632","snapshot_observed_at":"2026-08-07T04:44:59.994525Z","title":"A survey on benchmarks of multimodal large language models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T04:44:59.994525Z"},"links":{"cited_paper":"/paper/2408.08632","citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:c06fea2b694b8a2c74c37bb6395ecf8fd386ebf1959bbf532f5198e78332676d","observation_id":"f51502c7-a87f-45f3-a521-4b4b84b61990","resolution":{"observed_at":"2026-08-07T04:44:59.994525Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:00.035361Z","title":"From representation to reasoning: Towards both evidence and commonsense reasoning for video question-answering","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:00.035361Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:3741dc9b93cde74bc2a04305ca1f5b9262f6e17f4ce9bcd6873e146a8289d610","observation_id":"1862ce33-1422-43e6-adc1-16eed78166f6","resolution":{"observed_at":"2026-08-07T04:45:00.035361Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.14895","last_updated":"2022-05-30T07:26:54Z","snapshot_observed_at":"2026-08-08T01:34:06.455716Z","submitted_at":"2022-05-30T07:26:54Z","title":"From Representation to Reasoning: Towards both Evidence and Commonsense Reasoning for Video Question-Answering","version":1},"cited_work":{"arxiv_id":"2205.14895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2205.14895","snapshot_observed_at":"2026-08-07T04:45:11.809284Z","title":"From Representation to Reasoning: Towards both Evidence and Commonsense Reasoning for Video Question-Answering","venue":"cs.CV","work_id":"1b52e90e-56b6-495f-91d7-f91ce2527e65","year":2022},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:00.126076Z"},"links":{"cited_paper":"/paper/2205.14895","citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:c7f306db22dda5ae7a7c6c82395d67184b9c8c30ed989ad50f396b2c8e485e0b","observation_id":"2304d145-2dee-4ba1-bb10-a509a377a609","resolution":{"observed_at":"2026-08-07T04:45:11.884246Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:00.213211Z","title":"Openeqa: Embodied question answering in the era of foundation models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:00.213211Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:d814a6c066fc975ac850ab231b24fbb9eb011efde528e3257cbf1ecf8960afc5","observation_id":"8808adf5-2c93-4427-b982-5191d536c889","resolution":{"observed_at":"2026-08-07T04:45:00.213211Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09126","last_updated":"2023-08-17T17:59:59Z","snapshot_observed_at":"2026-07-06T16:07:21.951225Z","submitted_at":"2023-08-17T17:59:59Z","title":"EgoSchema: A Diagnostic Benchmark for Very Long-form Video Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09126","snapshot_observed_at":"2026-08-07T04:45:00.265342Z","title":"Egoschema: A diagnostic benchmark for very long-form video language understanding, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:00.265342Z"},"links":{"cited_paper":"/paper/2308.09126","citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:c8296722015d41051dc78c2d7c03a5c253a925a4958b9d3d9b3af17e52779c65","observation_id":"9e01ba51-5333-48b2-be59-6a13003a0f9b","resolution":{"observed_at":"2026-08-07T04:45:00.265342Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:00.369247Z","title":"Towards world simulator: Crafting physical commonsense-based benchmark for video generation,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:00.369247Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:f97beade57bccfde61ab3cba4090f44e2e6b145abc5e94990b2631f57dda92e9","observation_id":"97c7bbcd-57c1-4590-97bf-70ca5138bc23","resolution":{"observed_at":"2026-08-07T04:45:00.369247Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09038","last_updated":"2025-02-27T15:10:51Z","snapshot_observed_at":"2026-08-02T17:18:33.867969Z","submitted_at":"2025-01-14T20:59:37Z","title":"Do generative video models understand physical principles?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.09038","snapshot_observed_at":"2026-08-07T04:45:00.498442Z","title":"Do generative video models understand physical principles?arXiv preprint arXiv:2501.09038, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:00.498442Z"},"links":{"cited_paper":"/paper/2501.09038","citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:0209941c5d06a9b817ac453b26c25ef7aef1c4ee5e809594b521823da08de74f","observation_id":"21e1d744-225b-4017-a579-a961ece04325","resolution":{"observed_at":"2026-08-07T04:45:00.498442Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:00.579484Z","title":"Gpt-4v(ision) system card, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:00.579484Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:7dbe3b180465c5198dc92179dfe9ee0bc89b2318ca37fa5f41c245ca0d591490","observation_id":"c9157285-01e5-4d63-b039-e7c11e47c7f0","resolution":{"observed_at":"2026-08-07T04:45:00.579484Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:00.670428Z","title":"Gpt-4o, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:00.670428Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:a6b032beedc7d6f9b4be3363b5d4ff2d9d195c99e2361c95f7b553f706b0616f","observation_id":"c1f1413c-6394-4c13-a2db-c7f941a041c3","resolution":{"observed_at":"2026-08-07T04:45:00.670428Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:00.716172Z","title":"CRIPP-VQA: Counterfactual reasoning about implicit physical properties via video question answering","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:00.716172Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:28d0ca745c983f417d542c2e982a9b18a9e361d9d73f774992f7f1eb04280810","observation_id":"749f9399-637e-46c4-8840-ca1198b634c9","resolution":{"observed_at":"2026-08-07T04:45:00.716172Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2211.03779","last_updated":"2022-11-07T18:55:26Z","snapshot_observed_at":"2026-08-02T20:59:23.545562Z","submitted_at":"2022-11-07T18:55:26Z","title":"CRIPP-VQA: Counterfactual Reasoning about Implicit Physical Properties via Video Question Answering","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.03779","snapshot_observed_at":"2026-08-07T04:45:00.785774Z","title":"Cripp-vqa: Counterfactual reasoning about implicit physical properties via video question answering, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:00.785774Z"},"links":{"cited_paper":"/paper/2211.03779","citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:8b7468600905c8e6ad80f55ba2d299a526eacf0b162bb3a780dee8f77a147ff2","observation_id":"a27c0377-d165-4628-8356-6620bb1b3c8a","resolution":{"observed_at":"2026-08-07T04:45:00.785774Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11483","last_updated":"2023-08-22T14:54:59Z","snapshot_observed_at":"2026-08-07T01:33:49.598772Z","submitted_at":"2023-08-22T14:54:59Z","title":"Large Language Models Sensitivity to The Order of Options in Multiple-Choice Questions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.11483","snapshot_observed_at":"2026-08-07T04:45:00.866380Z","title":"Large language models sensitivity to the order of options in multiple-choice questions, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:00.866380Z"},"links":{"cited_paper":"/paper/2308.11483","citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:e44336852aff5e3c26e776ec824243602c8d9f5d50ac8434b59a677a825972a0","observation_id":"03b14deb-1181-4f56-8ae8-6fc0a0ffb1ec","resolution":{"observed_at":"2026-08-07T04:45:00.866380Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.13786","last_updated":"2023-10-30T18:35:48Z","snapshot_observed_at":"2026-07-06T15:31:13.189124Z","submitted_at":"2023-05-23T07:54:37Z","title":"Perception Test: A Diagnostic Benchmark for Multimodal Video Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.13786","snapshot_observed_at":"2026-08-07T04:45:01.031903Z","title":"Perception test: A diagnostic benchmark for multimodal video models, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:01.031903Z"},"links":{"cited_paper":"/paper/2305.13786","citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:7c48f0d7e526c09398f0953b7ca728ada97456ffa0705ef4c85ccfcc36c8f051","observation_id":"4a4ee60c-492e-4d88-a047-99ce02a09dfd","resolution":{"observed_at":"2026-08-07T04:45:01.031903Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.08813","last_updated":"2024-10-21T03:08:08Z","snapshot_observed_at":"2026-08-07T17:42:30.374808Z","submitted_at":"2024-05-14T17:59:02Z","title":"CinePile: A Long Video Question Answering Dataset and Benchmark","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.08813","snapshot_observed_at":"2026-08-07T04:45:01.238009Z","title":"Cinepile: A long video question answering dataset and benchmark, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:01.238009Z"},"links":{"cited_paper":"/paper/2405.08813","citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:19c0ba815c9ddfa57cdb4994138e1f81140aa45b2e7eeef21e023e037b61df2f","observation_id":"e967bfc9-8471-4deb-8a98-736eb0df26c4","resolution":{"observed_at":"2026-08-07T04:45:01.238009Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1803.07616","last_updated":"2020-02-11T10:05:20Z","snapshot_observed_at":"2026-08-09T00:07:58.496687Z","submitted_at":"2018-03-20T19:29:46Z","title":"IntPhys: A Framework and Benchmark for Visual Intuitive Physics Reasoning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1803.07616","snapshot_observed_at":"2026-08-07T04:45:01.399984Z","title":"Intphys: A framework and benchmark for visual intuitive physics reasoning, 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:01.399984Z"},"links":{"cited_paper":"/paper/1803.07616","citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:a70fcce9ce47516c2f95d5fce391b96d6e5db48e1849ae9d662f64b956271063","observation_id":"c8c37842-f559-41df-bc41-be92c2520e31","resolution":{"observed_at":"2026-08-07T04:45:01.399984Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:01.540151Z","title":"Ego4d goal-step: Toward hierarchical understanding of procedural activities","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:01.540151Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:3d74c30ad770cc02c7b16f8a3c35a2f69802971ebf6d86814bbbc34009fe7d94","observation_id":"20a78be1-ebcc-42d0-a548-475e26a304ad","resolution":{"observed_at":"2026-08-07T04:45:01.540151Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.07058","last_updated":"2022-03-11T19:40:26Z","snapshot_observed_at":"2026-07-06T11:57:42.646453Z","submitted_at":"2021-10-13T22:19:32Z","title":"Ego4D: Around the World in 3,000 Hours of Egocentric Video","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.07058","snapshot_observed_at":"2026-08-07T04:45:01.691636Z","title":"Ego4d: Around the world in 3,000 hours of egocentric video, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:01.691636Z"},"links":{"cited_paper":"/paper/2110.07058","citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:b28c2065a24fb1f94e9f5f0c656c815c9288490769d7f4bbe4b5a113333f815a","observation_id":"9d7017b1-3802-4473-a291-acd1e235caa7","resolution":{"observed_at":"2026-08-07T04:45:01.691636Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:01.841389Z","title":"Ego-exo4d: Understanding skilled human activity from first- and third-person perspectives,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:01.841389Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:e5ba68a43ef4a387f7af133ddd7966c3da3c05e60c909f3714add6a23d966418","observation_id":"c44a31b7-ccbc-4db4-861a-8726caa863b0","resolution":{"observed_at":"2026-08-07T04:45:01.841389Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05530","last_updated":"2024-12-16T17:39:39Z","snapshot_observed_at":"2026-07-06T17:41:42.995949Z","submitted_at":"2024-03-08T18:54:20Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05530","snapshot_observed_at":"2026-08-07T04:45:02.272938Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:02.272938Z"},"links":{"cited_paper":"/paper/2403.05530","citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:7b88ed5d7b5ade6932861c4dbbe7c9da9157fa03d36ead27fd983e92b3c90ec4","observation_id":"d8e0d45d-3866-4016-9d08-005fd97c6353","resolution":{"observed_at":"2026-08-07T04:45:02.272938Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:02.494976Z","title":"Gemini 2.5 flash, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:02.494976Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:c5e9072f3d957dc8ac7ca9fddc20142dcee2ee36490f9550f06ae3ad647c8394","observation_id":"702f372f-5f20-413c-9ae8-802e6337d664","resolution":{"observed_at":"2026-08-07T04:45:02.494976Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.18259","last_updated":"2024-09-25T21:55:38Z","snapshot_observed_at":"2026-08-08T17:40:49.067743Z","submitted_at":"2023-11-30T05:21:07Z","title":"Ego-Exo4D: Understanding Skilled Human Activity from First- and Third-Person Perspectives","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.18259","snapshot_observed_at":"2026-08-07T04:45:02.057866Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:02.057866Z"},"links":{"cited_paper":"/paper/2311.18259","citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:fb1424b2237a2618d7964af2ad28734705d194c3f6826ae339a14c5fa71d24f9","observation_id":"97ddf7df-e792-4d24-bb47-75b20b0a4dc9","resolution":{"observed_at":"2026-08-07T04:45:02.057866Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.01620","last_updated":"2023-11-02T22:17:03Z","snapshot_observed_at":"2026-08-05T18:01:04.300201Z","submitted_at":"2023-11-02T22:17:03Z","title":"ACQUIRED: A Dataset for Answering Counterfactual Questions In Real-Life Videos","version":1},"cited_work":{"arxiv_id":"2311.01620","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.01620","snapshot_observed_at":"2026-08-07T04:45:11.493440Z","title":"ACQUIRED: A Dataset for Answering Counterfactual Questions In Real-Life Videos","venue":"cs.CV","work_id":"58e7b764-3fe8-4ed4-b63f-e250bc60e04a","year":2023},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:02.802410Z"},"links":{"cited_paper":"/paper/2311.01620","citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:664ed06b560d7db29d4e4caf192cc0c6f47313bea7d3ed88c1b557fcda54fb4d","observation_id":"f1e9217a-6459-4f47-b41b-385d3079dcfb","resolution":{"observed_at":"2026-08-07T04:45:11.543741Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2105.08276","last_updated":"2021-05-23T09:37:24Z","snapshot_observed_at":"2026-07-06T11:10:20.716791Z","submitted_at":"2021-05-18T04:56:46Z","title":"NExT-QA:Next Phase of Question-Answering to Explaining Temporal Actions","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2105.08276","snapshot_observed_at":"2026-08-07T04:45:02.947046Z","title":"Next-qa:next phase of question-answering to explaining temporal actions, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:02.947046Z"},"links":{"cited_paper":"/paper/2105.08276","citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:9bc10550f5127b18a6623f477f7b51e16bece3cd1fa4fd137e9dfdf5fa193a10","observation_id":"c804413e-8a94-4766-bccc-da5a0043e225","resolution":{"observed_at":"2026-08-07T04:45:02.947046Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-07T04:45:02.651654Z","title":"The llama 3 herd of models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:02.651654Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:a2235e822185a6fd095996747ac6960297ec6624e540f2a0a93324a0663a1fcd","observation_id":"c64d2533-aad9-421e-a88f-b4eb22396ec5","resolution":{"observed_at":"2026-08-07T04:45:02.651654Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1906.02467","last_updated":"2019-06-06T08:08:14Z","snapshot_observed_at":"2026-07-06T07:58:24.206559Z","submitted_at":"2019-06-06T08:08:14Z","title":"ActivityNet-QA: A Dataset for Understanding Complex Web Videos via Question Answering","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1906.02467","snapshot_observed_at":"2026-08-07T04:45:03.277823Z","title":"Activitynet-qa: A dataset for understanding complex web videos via question answering, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:03.277823Z"},"links":{"cited_paper":"/paper/1906.02467","citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:becee8db7e15d2476842570f35dff4a348c18384e5e8ed9769604ea1b1f8f129","observation_id":"b8fc14fe-d142-47cc-a5c1-094628cdc9ec","resolution":{"observed_at":"2026-08-07T04:45:03.277823Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.13601","last_updated":"2024-05-28T05:36:23Z","snapshot_observed_at":"2026-08-08T11:20:34.553955Z","submitted_at":"2024-01-24T17:10:45Z","title":"MM-LLMs: Recent Advances in MultiModal Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.13601","snapshot_observed_at":"2026-08-07T04:45:03.370411Z","title":"Mm-llms: Recent advances in multimodal large language models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:03.370411Z"},"links":{"cited_paper":"/paper/2401.13601","citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:b3b0fd308f17e10c9fa27865d7f58e6a8b1ad0663f66465348f9b5026ce9ed9e","observation_id":"aa3d421f-66b2-4e32-8b7e-fc29c7556eed","resolution":{"observed_at":"2026-08-07T04:45:03.370411Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1910.01442","last_updated":"2020-03-08T00:09:07Z","snapshot_observed_at":"2026-07-06T08:26:38.349660Z","submitted_at":"2019-10-03T13:16:36Z","title":"CLEVRER: CoLlision Events for Video REpresentation and Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1910.01442","snapshot_observed_at":"2026-08-07T04:45:03.139320Z","title":"Tenenbaum","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:03.139320Z"},"links":{"cited_paper":"/paper/1910.01442","citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:3bd64e93f706ce9d7cac4788f4a041696c68764c24c9967328fe566da4257214","observation_id":"923b733d-9791-40d0-8884-de6d6ac0baa3","resolution":{"observed_at":"2026-08-07T04:45:03.139320Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.01225","last_updated":"2022-11-02T05:25:06Z","snapshot_observed_at":"2026-07-30T11:11:52.165985Z","submitted_at":"2022-03-02T16:34:09Z","title":"Video Question Answering: Datasets, Algorithms and Challenges","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.01225","snapshot_observed_at":"2026-08-07T04:45:03.529058Z","title":"distractors","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:03.529058Z"},"links":{"cited_paper":"/paper/2203.01225","citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:90c26bc95070611782bee3191ab8d14ecdf8206fceeb3056f053e627116f3e62","observation_id":"adce09e7-20c2-4d1d-bc7b-33b0047aae42","resolution":{"observed_at":"2026-08-07T04:45:03.529058Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:03.465818Z","title":"Contphy: Continuum physical concept learning and reasoning from videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:03.465818Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:015ec195356251f6ea93270d7100037039fa63363a84a9e32b92b26a7f89159c","observation_id":"ba2d123e-af32-432e-9b9a-1b00a77f2ba6","resolution":{"observed_at":"2026-08-07T04:45:03.465818Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:03.623339Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:03.623339Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:63f66136223fbd7ef64cdc1efd55cb9e6aec5b3458317c4984dcd74501f6257b","observation_id":"3e0a10d5-3cdb-43a2-8942-8d21c0d6ac0f","resolution":{"observed_at":"2026-08-07T04:45:03.623339Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:03.673237Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:03.673237Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:162db82ddfde9394ba36dc10cb295e708ddf4d9d10cbb61d6161c4f271f3cb1e","observation_id":"b455ab8f-6da4-4692-92c2-7124f56d40f6","resolution":{"observed_at":"2026-08-07T04:45:03.673237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:03.795621Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:03.795621Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:97134bf2caa8ad238deda3f3630d441d58de78a04df0ba5e8b5e76c88bc1ce98","observation_id":"b3a22ddb-005e-41d1-97b3-054150954b80","resolution":{"observed_at":"2026-08-07T04:45:03.795621Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:03.883579Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:03.883579Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:b05c77bad52adf395ee342d93bc6d1c721c064223fee3856dec64064b3c37a47","observation_id":"ce646c3f-214e-412f-aab7-e5ba61f4bd2c","resolution":{"observed_at":"2026-08-07T04:45:03.883579Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:04.005637Z","title":"Descriptive ii","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:04.005637Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:de875de2deaed9455b9306f6c0dee855f942097759b04f6dfb0643c553834a73","observation_id":"6d430c7b-a152-478e-91a7-97cdcdc9e2c6","resolution":{"observed_at":"2026-08-07T04:45:04.005637Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:04.103939Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:04.103939Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:11e2a134be0f0a5c88132e8c1d49ead2fba65fcf71a737a3aff472f473903234","observation_id":"0a19e898-046f-472b-8eb8-6152e0d9fe4b","resolution":{"observed_at":"2026-08-07T04:45:04.103939Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:04.222322Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:04.222322Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:f55875563ecf8aa0a2461b056dfe7b5f819365b538232cb818252ac7d48e7ee4","observation_id":"caa9e58d-8780-4274-a088-8f0408491a5d","resolution":{"observed_at":"2026-08-07T04:45:04.222322Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:19.395552Z","title":null,"venue":null,"work_id":"e773cd05-c048-47d6-a742-85feedc9d264","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:04.318714Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:e4a88e10f595de910de2f8f9efc641258f60fadb4597867440a3e55732ebb3eb","observation_id":"b878ef8a-bc9e-43b5-9067-6e9f7a56307a","resolution":{"observed_at":"2026-08-07T04:45:19.399783Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:19.381252Z","title":null,"venue":null,"work_id":"07c0277c-9470-44f8-b94f-30bebb18a066","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:04.377784Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:71a77a4c3cfea80a5c1659a3ba96c696b8006a02c79bb673ffc0565bcf68fd04","observation_id":"9ed162d3-46c5-4387-b89c-6e3864f4db40","resolution":{"observed_at":"2026-08-07T04:45:19.385874Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:19.367542Z","title":"your answer","venue":null,"work_id":"82c2ef23-fe2b-42a4-96ef-474ec59f8134","year":2025},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:04.472564Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:b5a3865b83f15104b2d57d7dd7962e64ba049bafc63ef15a59f53899363693c5","observation_id":"bfc1e739-9284-43c3-86fa-0d16861d3377","resolution":{"observed_at":"2026-08-07T04:45:19.372089Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:19.339861Z","title":null,"venue":null,"work_id":"e105d4ad-0166-4deb-a0d1-785543c534cd","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:04.661137Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:f3fbf96834ae60137ae375f2f46b8a0975826ae2fd18ce00b33118f68f08171c","observation_id":"e8592a99-dfa7-4c51-997c-f037dbdf22f4","resolution":{"observed_at":"2026-08-07T04:45:19.344063Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:19.312922Z","title":"The answers should extremely plausible, but incorrect, based on what you see in the Clip","venue":null,"work_id":"8397196f-7072-48ed-9df7-61c4ec874764","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:04.843854Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:f0c8c808e937e3089a3420576439f77eae7e506f8c1f6bfa4f39f11933d64686","observation_id":"08e2f3b2-ab1b-442b-900a-986610d5ff98","resolution":{"observed_at":"2026-08-07T04:45:19.317383Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:19.186478Z","title":"Distractors **must** be similar in length and complexity to the Revised Answer","venue":null,"work_id":"6f36b39a-b960-4e0e-9dc5-b62f3494fc26","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:04.943142Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:248d988d086bf050f55dbe3b48951080cdce2a64d042d7080449500a47099bf5","observation_id":"13603173-c799-4e6e-9083-8c0cd9bff05c","resolution":{"observed_at":"2026-08-07T04:45:19.282688Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:19.080214Z","title":"Revise the Distractors as needed unil you have your list of four","venue":null,"work_id":"4d08be94-c9f8-4fc4-8cef-214ec50903da","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:05.030027Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:f380c6fcf3a5362df7bb283ac330ca1d7de338c9c71dd395f8e346368c84ffea","observation_id":"cb9ffe59-0a9d-4a10-b76d-f49ee9c8f95d","resolution":{"observed_at":"2026-08-07T04:45:19.132481Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:18.947344Z","title":"Modify the Distractors as needed to ensure they are all different from the Revised Answer","venue":null,"work_id":"91493c4a-660e-4aaa-9b99-4bf936495bc9","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:05.105756Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:35efcda0a1e64641b4005275078f435146c1ab46aaf38bf5ecab657d7801ec30","observation_id":"e0d14546-bd10-4737-b420-aa4e85dd3b8b","resolution":{"observed_at":"2026-08-07T04:45:19.003840Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:18.810001Z","title":"[Distractor 1| Distractor 2| etc...]","venue":null,"work_id":"53ec659a-a9e7-444a-ad79-b7f125d055ca","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:05.228857Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:2b58dd472b3862484d00cf6a4b7c7c57b01ed5573941b8ffcfdea1ebbdad0fce","observation_id":"a62e7faf-e548-4a2e-8ea5-96acdb4eab23","resolution":{"observed_at":"2026-08-07T04:45:18.861804Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:18.672910Z","title":null,"venue":null,"work_id":"2166198d-de96-45ac-8880-ed99f644ccb0","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:05.330856Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:66fc10858ed399a5d7bcc4473ae949235cb25fdf16f3b910f4705e3d2ec8c1c4","observation_id":"dc21491a-ec60-42ad-afcd-f644f6c6bad6","resolution":{"observed_at":"2026-08-07T04:45:18.727088Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:18.616451Z","title":null,"venue":null,"work_id":"1024c5bf-7749-4d75-99f0-4173ea3ba1e6","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:05.427491Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:9cfb8325a93a1d6bc2717f81718b2ef2746b36ece4ba88a47da1dd35ad2f8c21","observation_id":"ea85bf75-757c-42ca-b1d0-0ef4b8007f58","resolution":{"observed_at":"2026-08-07T04:45:18.647126Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:18.442322Z","title":"[’distractor 1.’| ’distractor 2.’| ’distractor 3.’| ’distractor 4.’]","venue":null,"work_id":"f7717ca6-bc8e-4afd-8162-a55a2b0a4607","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:05.523172Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:0f2ebfc298a5f0348dc2b41407221b3b8bdd92332bf960ac9bbdae9c4e17e745","observation_id":"5f325754-d349-4d34-afc5-ac791a089997","resolution":{"observed_at":"2026-08-07T04:45:18.524783Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:17.989879Z","title":"You follow the following steps in this task:","venue":null,"work_id":"16614e27-77ff-4bff-8145-6791edc551f1","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:06.751190Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:be6054d0b6ad2d23fe44c5eada14feb987a4355ddc7adbb9730b2e4f34ddf156","observation_id":"ae265068-59e1-4ebc-9d0b-d60ad4e025f3","resolution":{"observed_at":"2026-08-07T04:45:18.007479Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:17.859482Z","title":null,"venue":null,"work_id":"404dfe0e-07e6-439e-bc18-20619c2dcf3d","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:06.830679Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:509007929582820c101275860ebe084cd43223e9e7488276b06b350464a52b59","observation_id":"8c44117c-ed4c-449e-953d-d6872a8bf572","resolution":{"observed_at":"2026-08-07T04:45:17.904872Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:17.588278Z","title":"Please sound as polished as possible and make certain to capture the gist of the Original Answer","venue":null,"work_id":"9b4adb9f-6837-423c-8351-373754cd849b","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:06.999865Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:d54d8083de84cb31b37c4807a49de413136df655dbda80c3da24dce339856b4e","observation_id":"417adc5e-726d-4942-b15e-f06abac1c6ac","resolution":{"observed_at":"2026-08-07T04:45:17.728621Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:17.363831Z","title":null,"venue":null,"work_id":"25a35831-c789-4056-a6d9-f9925f7d0678","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:07.087752Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:05ffd352bdffbf53d123e2e27894885b97d7bfa1b0c07d8b8016ccd9a39686fb","observation_id":"776bfcc8-8a41-4d95-b139-aa8bb98f6543","resolution":{"observed_at":"2026-08-07T04:45:17.460432Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:17.293238Z","title":null,"venue":null,"work_id":"02b0dcd5-36ff-4a94-9364-b78d7a527287","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:07.218298Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:e63126ac5c65f71a225f9f44ed8774f80729ef9b29d19be1ce20ec08c1f0136c","observation_id":"6abad4e7-1ec6-4c54-bf90-c574063611a3","resolution":{"observed_at":"2026-08-07T04:45:17.351485Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:17.105305Z","title":"Ensure that language in the question reflects the language in the rewritten answer","venue":null,"work_id":"ea1ea69a-8ab6-49dc-9128-a887e4520a76","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:07.360635Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:659b547972b8f36feaeb857b0777788c8244fac5dccbe772fd44450ed6583523","observation_id":"25b499ec-b762-41b2-9d47-f114f1860852","resolution":{"observed_at":"2026-08-07T04:45:17.173730Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:16.925330Z","title":"Now make certain that the rewritten question has the same meaning as the Original Question","venue":null,"work_id":"9c2378e0-6ebb-472d-9391-345541eefd06","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:07.501348Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:c58a7e4ab3307541331ef484b9e25c55c9b42d1a0a7612a197ffda3db15e54de","observation_id":"e67ae395-8e56-4fa1-a558-efd4ab5fc516","resolution":{"observed_at":"2026-08-07T04:45:17.013439Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:16.745031Z","title":"the rewritten answer you came up with in step 2, adjusted for any changes from steps 3 and 4","venue":null,"work_id":"4c71287b-8393-4c0d-ac11-aab008af4ebf","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:07.585936Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:80c78bed014f75108f88c61264c4e0d7f81347520c7b86a2aa07aaf2a0b44426","observation_id":"d1d5d2d8-cb61-441b-9477-d5f366fa7f36","resolution":{"observed_at":"2026-08-07T04:45:16.836892Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:16.642335Z","title":null,"venue":null,"work_id":"677e3dfb-0af6-48cc-ba37-5fa77f0a1964","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:07.676622Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:57c9292451a0459c4b0517d3f1548d928c3d51ae764de0c3087ed01e201e2f49","observation_id":"e62c39c6-c892-49d7-aa02-e69d233ed2b9","resolution":{"observed_at":"2026-08-07T04:45:16.688247Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:16.453432Z","title":null,"venue":null,"work_id":"12450c00-071d-4a40-9258-ebfce95930d9","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:07.920648Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:fd6513f4b0a22657a6e7d033394497afbcf73dac026f924bb62f18f70dbd32e6","observation_id":"f31a9cca-b62e-4f13-9dd9-e0409acbbf54","resolution":{"observed_at":"2026-08-07T04:45:16.545485Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:16.261575Z","title":null,"venue":null,"work_id":"ef12c1bb-e95f-46bb-825e-c9448c7cd4d6","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:08.237140Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:5fce902bce048cfdbec06c0a98a318b4eafa86a9f4d9af34e68728d2a43e047d","observation_id":"839f3e0f-ed82-4901-8089-db7c390c3cca","resolution":{"observed_at":"2026-08-07T04:45:16.342273Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:16.081410Z","title":"the rewritten New Answer you came up with in step 2","venue":null,"work_id":"ecdb6e89-3989-4be5-8fe2-9d844bdece15","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:08.453484Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:ee5e763d98453ced7566bbb4bdee350f8133e51e0a30fc4a5e4b1db4799816da","observation_id":"90325542-f486-4c0c-b365-8794c509e2b9","resolution":{"observed_at":"2026-08-07T04:45:16.160603Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:15.912285Z","title":"Important Words will only be found in the Original Question","venue":null,"work_id":"4061ddbc-be02-41cf-9a82-9e9a275d812b","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:08.734053Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:b3416191e55eb9e28fb60623a1b1b5ea5fcfef8a2725c4cc693ac5d2a299819a","observation_id":"048b3a48-a591-42e5-b863-a24c12783ce3","resolution":{"observed_at":"2026-08-07T04:45:15.992354Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:15.630266Z","title":null,"venue":null,"work_id":"0d663bf1-412c-4316-9249-2dcb4a83d04b","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:08.848910Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:5579c0001afa5bc269e67ca501143d1192a34789b6a92faabd5093fc2443c267","observation_id":"b7a6d53e-e447-48f8-80a1-b9404b4f3645","resolution":{"observed_at":"2026-08-07T04:45:15.804421Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:15.333493Z","title":"Important Words will only be found in the Original Question","venue":null,"work_id":"8888acce-073b-4076-93d5-03ac5ad3a73c","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:08.971112Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:79150c212744f62b1d0c9a4f3001a36aa7acd18dd47141fa037b2b413a9a6fd2","observation_id":"bb6ea2bd-741f-4451-a07a-225a1f754528","resolution":{"observed_at":"2026-08-07T04:45:15.469377Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:14.980285Z","title":"defensive position","venue":null,"work_id":"08a7119d-03c7-4951-8b08-70322cc0e804","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:09.100488Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:0606876d0d3ae5947337ae0e9a86c846537ab623c5c6e0b7213ef54aa9add82e","observation_id":"378e9220-5ea6-49d2-8dc9-74aff1d76a74","resolution":{"observed_at":"2026-08-07T04:45:15.142199Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:14.529987Z","title":null,"venue":null,"work_id":"9cd64fc6-c375-4997-839f-1d7247983e5a","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:09.173312Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:865bf6af3cd7b6924098acfb71f2fbc8c534b30834606e238615697fc5a96fab","observation_id":"e99811ad-d18c-4bb0-82d3-c20636c1cd4b","resolution":{"observed_at":"2026-08-07T04:45:14.719451Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:14.211343Z","title":null,"venue":null,"work_id":"f6743451-f208-41cd-a475-07134c5ddac0","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:09.253384Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:86bc3db1b0b37e762b82c72da6cadbedbcc7801158a1221b359f606631f41d7a","observation_id":"5f126798-fd9e-48a2-9e68-eee431f9b6e9","resolution":{"observed_at":"2026-08-07T04:45:14.311917Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:14.093059Z","title":"the answer you came up with in step 1, 2, 3, 4, and 6","venue":null,"work_id":"b647210d-96a9-4706-ad2c-18f2f0fc0d81","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:09.362150Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:f52575ba8bc9253d5166e5af3b5d10277bc90e82ed1799c737c4e1f9217426ab","observation_id":"8829dbe9-36c6-4ae8-acb2-17e34c766c6c","resolution":{"observed_at":"2026-08-07T04:45:14.149284Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:19.353569Z","title":null,"venue":null,"work_id":"32c1156a-2a9f-448d-8cab-45a1b50ac538","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:09.450483Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:56fc9f2bf8480a5b4576a5db37732463beb3b5e3ce77a54b5c675d09a1939ccc","observation_id":"72e60af5-8fbe-4dd5-92e4-b6983a202ea7","resolution":{"observed_at":"2026-08-07T04:45:19.357593Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:13.961445Z","title":"|\", \",\",","venue":null,"work_id":"380a852a-ec39-4993-92d6-20cb7d3e7952","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:09.527449Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:fc63eaa21bb7a5a0426d5c0b3f0e2c072a962ef3267d0539e002feb8ffa14dea","observation_id":"347a4932-5826-4a17-a9b2-4b280df6f5db","resolution":{"observed_at":"2026-08-07T04:45:14.033219Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:19.326359Z","title":null,"venue":null,"work_id":"1c95e466-ac9d-4667-a293-d20f983876d9","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:09.618527Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:a6934477aab059a35ef0c0847c113b586a94238b2a8be1688f8fe4f03732a6d0","observation_id":"2e6c783a-28fd-41a7-9481-5bc075b63380","resolution":{"observed_at":"2026-08-07T04:45:19.330274Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:13.817982Z","title":"They **must** be similar in length and complexity to the Revised Answer","venue":null,"work_id":"a4851dc9-4566-40d8-8afa-10714d5b4c3e","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:09.736526Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:e3372665d7edc91d62dca330426532f372e97ee4846da4e67d400e925a13d406","observation_id":"d73db59d-0170-4689-925e-6eb7f08ccbf3","resolution":{"observed_at":"2026-08-07T04:45:13.888407Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:13.686562Z","title":"Encourage yourself to think outside the box and consider unconventional explanations or scenarios","venue":null,"work_id":"36fe417c-48ed-4002-b97e-3d8d53f998bf","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:09.840518Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:e90cad7ae41ad4f047bf27078506fa5e3ab33ef83c9c7d68fa954c6c49730df6","observation_id":"faa44f19-5779-4bdb-93af-27f84746212f","resolution":{"observed_at":"2026-08-07T04:45:13.744472Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:13.532651Z","title":"**Never** repeate the Revised Answer in the New Distractors","venue":null,"work_id":"25aeeb35-2072-4f8b-8284-0cdd86f75d12","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:09.927199Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:aa55dd774bc68eb69a78f65cf5e9c50d42b798427d08df2aab03de1bc2c60cbe","observation_id":"6f525a2b-22b1-44aa-979d-bcd6464bf869","resolution":{"observed_at":"2026-08-07T04:45:13.595444Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:13.409706Z","title":null,"venue":null,"work_id":"487dffff-bda6-4ea4-841d-2f1b0cf88e37","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:09.995148Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:3b45ff3ac77021242b56a960da8abdbe589f8ec60a267e7e4c505782b7763647","observation_id":"2d0cae44-f069-433d-85e3-c54ff0ef6dc3","resolution":{"observed_at":"2026-08-07T04:45:13.458271Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:13.194664Z","title":"[New Distractor 1| New Distractor 2| etc...]","venue":null,"work_id":"1546737a-d3dd-4ace-9f2f-5a23a7ce2a5f","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:10.078385Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:42ca218fd592a57bd1e4ca58a3d3b89aa0bd5f490b830168e1ec6efec47e6f8f","observation_id":"eb5743e3-a05c-4295-9b4a-a15c187b402e","resolution":{"observed_at":"2026-08-07T04:45:13.336757Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:18.268323Z","title":null,"venue":null,"work_id":"4ac592d0-114e-409a-9e50-fc10b427be97","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":101,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:10.176340Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:d3ff1a0db387e7fecdf6b24ac21a71e1f6c9302e590f782649bcb4ef077981c6","observation_id":"92b41005-d995-44f6-aa4e-25ee9fb60875","resolution":{"observed_at":"2026-08-07T04:45:18.328745Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:18.137707Z","title":null,"venue":null,"work_id":"34c92f7f-543e-4b5d-a6a4-79f92ad67c4e","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":102,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:10.247590Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:388d12bf7b9cbc7fa09cc84e70c9664f7ed997946dbb9bc521dac4dc1fdb93a6","observation_id":"8e6999b8-a2d1-4a70-ae5c-bcfd2d8f8f8b","resolution":{"observed_at":"2026-08-07T04:45:18.197808Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:18.049457Z","title":null,"venue":null,"work_id":"d1eb353d-33a1-49e5-a312-2b1c5b8cd030","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":103,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:10.338069Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:5bd70366b71b7a2e9a68238418ee94bf316c06108201d16d0e789b4d96c7ce4c","observation_id":"2ecd64e7-a254-4171-9491-740e6a9b8feb","resolution":{"observed_at":"2026-08-07T04:45:18.063729Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:12.885462Z","title":"The person will move backward to create space B","venue":null,"work_id":"4ec70636-fcbc-46f2-bb10-1c7d6724617c","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":104,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:10.446201Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:e0f9d7bb9ea53a8dbe382e96a10d840ad0201c234f77e96ee345784fee272d8a","observation_id":"5e226ed2-0529-490f-9a50-76747d0ebb61","resolution":{"observed_at":"2026-08-07T04:45:13.010551Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:12.619128Z","title":"It is supplied as a string of numbered answers separated by a pipe |","venue":null,"work_id":"dfa3f991-c14c-4aa6-87dc-42601bd69ea8","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":105,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:10.544142Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:14ec05d7d7a9dd13062a55b25135a1e62a3eb62ee8721dce2799f23f6d4969b1","observation_id":"1bc72356-c2cb-408e-9b29-3fdf391d120c","resolution":{"observed_at":"2026-08-07T04:45:12.772062Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:12.476557Z","title":"You must choose an answer, even if you are not sure","venue":null,"work_id":"c194bf04-10d7-4a2f-af2b-59a0b8f6f439","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":106,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:10.672887Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:34771ea17a5f0a803be476fba4abff254abd31cff229fc22ee171ff9e2267342","observation_id":"46d4a69b-4f3a-47f0-b52d-bb61c363245d","resolution":{"observed_at":"2026-08-07T04:45:12.538689Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"8264.5559","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:45:11.161044Z","title":"Your Answer from the Possible Answers as a string","venue":null,"work_id":"1a656ffe-a283-4141-b2e4-c11182e92db5","year":null},"citing_paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","version":1},"reference_index":107,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:10.764292Z"},"links":{"citing_paper":"/paper/2506.09943"},"observation_digest":"sha256:22bad668a15c76028b89535e5687fb44d0db289c1d1999b8551c86def2b992fd","observation_id":"d14643e4-204c-46b9-907c-e414a8154e82","resolution":{"observed_at":"2026-08-07T04:45:11.216959Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.09943","last_updated":"2025-06-11T17:10:36Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-08T16:25:33.048853Z","submitted_at":"2025-06-11T17:10:36Z","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":1,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":72,"verified_exact":1,"verified_fuzzy":25},"total_outbound_references":104},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 100 of 104 outbound references and 18 inbound Pith citation observations for arXiv:2506.09943."}