{"as_of":"2026-08-05T14:46:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:d1aecc8b9c8f89caabc688e2160e4553f02268bee387113ca4200c13c5a02cb8","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":33,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":33,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":33,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":33,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-03T21:14:08.671259Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-09T18:06:25.898385Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","version":2},"cited_work":{"arxiv_id":"2501.16411","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.16411","snapshot_observed_at":"2026-07-09T18:06:25.898385Z","title":"PhysBench: Benchmarking and enhancing vision-language models for physical world understanding","venue":"cs.CV","work_id":"3a5e1663-11b8-43aa-8e78-33ab23326881","year":2025},"citing_paper":{"arxiv_id":"2511.10946","last_updated":"2026-04-15T01:06:55Z","snapshot_observed_at":"2026-08-03T12:49:21.604140Z","submitted_at":"2025-11-14T04:16:09Z","title":"Abstract 3D Perception for Spatial Intelligence in Vision-Language Models","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-17T22:43:16.761970Z"},"links":{"cited_paper":"/paper/2501.16411","citing_paper":"/paper/2511.10946"},"observation_digest":"sha256:69c88ea06aba6f6f4e56fca26747c220011c9bfa16ef68cb5f67851708880a80","observation_id":"d5e2afbd-c8a4-42e7-98b2-3df09b155d46","resolution":{"observed_at":"2026-05-17T22:45:24.431565Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.16411","snapshot_observed_at":"2026-08-03T21:14:08.671259Z","title":"Physbench: Bench- marking and enhancing vision-language models for physical world understanding.arXiv preprint arXiv:2501.16411, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2511.17649","last_updated":"2026-07-07T04:04:31Z","snapshot_observed_at":"2026-08-03T21:14:07.753245Z","submitted_at":"2025-11-20T09:52:20Z","title":"SWITCH: Benchmarking Modeling and Handling of Tangible Interfaces in Long-horizon Embodied Scenarios","version":5},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-03T21:14:08.671259Z"},"links":{"cited_paper":"/paper/2501.16411","citing_paper":"/paper/2511.17649"},"observation_digest":"sha256:ca87a70fc58480b2b2b25b97a5f486e31aff10bf8ba1248f092332a4d04a6cf1","observation_id":"78f07fa2-0a71-46e7-ab1d-4b349425d388","resolution":{"observed_at":"2026-08-03T21:14:08.671259Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","version":2},"cited_work":{"arxiv_id":"2501.16411","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.16411","snapshot_observed_at":"2026-07-09T18:06:25.898385Z","title":"PhysBench: Benchmarking and enhancing vision-language models for physical world understanding","venue":"cs.CV","work_id":"3a5e1663-11b8-43aa-8e78-33ab23326881","year":2025},"citing_paper":{"arxiv_id":"2511.18373","last_updated":"2026-04-11T05:44:20Z","snapshot_observed_at":"2026-07-06T22:36:44.887709Z","submitted_at":"2025-11-23T09:43:44Z","title":"MASS: Motion-Aware Spatial-Temporal Grounding for Physics Reasoning and Comprehension in Vision-Language Models","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-17T05:55:11.495430Z"},"links":{"cited_paper":"/paper/2501.16411","citing_paper":"/paper/2511.18373"},"observation_digest":"sha256:2624f3212415a84200a9ab5346df429fe137f455558655f9569d9f3c53c01b4a","observation_id":"466b7f2c-449d-402f-b6c3-6a4b6c17a26f","resolution":{"observed_at":"2026-05-17T05:59:08.633140Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.16411","snapshot_observed_at":"2026-08-03T17:17:05.467145Z","title":"Physbench: Benchmarking and enhancing vision-language models for physical world under- standing.arXiv preprint arXiv:2501.16411, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.10342","last_updated":"2026-06-26T22:35:57Z","snapshot_observed_at":"2026-08-03T17:17:02.803333Z","submitted_at":"2025-12-11T06:46:51Z","title":"CoSPlan: Corrective Sequential Planning via Scene Graph Incremental Updates","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-03T17:17:05.467145Z"},"links":{"cited_paper":"/paper/2501.16411","citing_paper":"/paper/2512.10342"},"observation_digest":"sha256:9328945c725fa1314178a2b71a955c0ddcab419ca45de9a8f8c63dca0411b3eb","observation_id":"88b3674e-346b-4206-a479-2edee83aa4e4","resolution":{"observed_at":"2026-08-03T17:17:05.467145Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.16411","snapshot_observed_at":"2026-08-03T16:47:24.462509Z","title":"Physbench: Benchmarking and enhancing vision-language models for physical world under- standing.arXiv preprint arXiv:2501.16411, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.11995","last_updated":"2026-06-09T17:48:42Z","snapshot_observed_at":"2026-08-03T21:58:57.150622Z","submitted_at":"2025-12-12T19:18:41Z","title":"V-REX: Benchmarking Exploratory Visual Reasoning via Chain-of-Questions","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-03T16:47:24.462509Z"},"links":{"cited_paper":"/paper/2501.16411","citing_paper":"/paper/2512.11995"},"observation_digest":"sha256:3220d6c0e02d9db03176c5f6bf56d3bb5ec368d25cc4fee1eb8d24b2e4b1a498","observation_id":"b379fa1c-68a0-47ce-829c-38684ac4ae24","resolution":{"observed_at":"2026-08-03T16:47:24.462509Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","version":2},"cited_work":{"arxiv_id":"2501.16411","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.16411","snapshot_observed_at":"2026-07-09T18:06:25.898385Z","title":"PhysBench: Benchmarking and enhancing vision-language models for physical world understanding","venue":"cs.CV","work_id":"3a5e1663-11b8-43aa-8e78-33ab23326881","year":2025},"citing_paper":{"arxiv_id":"2512.23292","last_updated":"2026-05-20T15:48:38Z","snapshot_observed_at":"2026-08-01T23:47:11.303976Z","submitted_at":"2025-12-29T08:26:27Z","title":"Agentic Physical AI toward a Domain-Specific Foundation Model for Nuclear Reactor Control","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-21T16:57:19.490074Z"},"links":{"cited_paper":"/paper/2501.16411","citing_paper":"/paper/2512.23292"},"observation_digest":"sha256:7fdd1f264636a08f81e5968a0c4a00c33b7f3d23731865cb9af9bbf5b98b5c9a","observation_id":"4ad7f2f2-4cb9-40be-a877-71b08816f498","resolution":{"observed_at":"2026-05-21T17:00:24.008404Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","version":2},"cited_work":{"arxiv_id":"2501.16411","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.16411","snapshot_observed_at":"2026-07-09T18:06:25.898385Z","title":"PhysBench: Benchmarking and enhancing vision-language models for physical world understanding","venue":"cs.CV","work_id":"3a5e1663-11b8-43aa-8e78-33ab23326881","year":2025},"citing_paper":{"arxiv_id":"2603.03944","last_updated":"2026-04-03T20:11:12Z","snapshot_observed_at":"2026-07-06T22:47:46.409064Z","submitted_at":"2026-03-04T11:09:39Z","title":"SCP: Spatial Causal Prediction in Video","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-15T16:47:44.523606Z"},"links":{"cited_paper":"/paper/2501.16411","citing_paper":"/paper/2603.03944"},"observation_digest":"sha256:8c572c454184a42d03a3909a4cd0b994af94e6db9d1ace03e3c48e51a95dd2fa","observation_id":"c8afffaa-b52f-4a36-8ee1-4d4446130432","resolution":{"observed_at":"2026-05-15T16:50:11.227229Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","version":2},"cited_work":{"arxiv_id":"2501.16411","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.16411","snapshot_observed_at":"2026-07-09T18:06:25.898385Z","title":"PhysBench: Benchmarking and enhancing vision-language models for physical world understanding","venue":"cs.CV","work_id":"3a5e1663-11b8-43aa-8e78-33ab23326881","year":2025},"citing_paper":{"arxiv_id":"2604.00799","last_updated":"2026-04-02T21:20:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-01T12:06:54Z","title":"Multimodal Language Models Cannot Spot Spatial Inconsistencies","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-05-13T23:12:27.333405Z"},"links":{"cited_paper":"/paper/2501.16411","citing_paper":"/paper/2604.00799"},"observation_digest":"sha256:1c9f9e6307f6520fe502864cbdc3911e1145356283f38c19985e5a5efffdcfe9","observation_id":"253d7d01-5785-4c20-8e07-4864a5d3dbbe","resolution":{"observed_at":"2026-05-13T23:13:24.768838Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","version":2},"cited_work":{"arxiv_id":"2501.16411","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.16411","snapshot_observed_at":"2026-07-09T18:06:25.898385Z","title":"PhysBench: Benchmarking and enhancing vision-language models for physical world understanding","venue":"cs.CV","work_id":"3a5e1663-11b8-43aa-8e78-33ab23326881","year":2025},"citing_paper":{"arxiv_id":"2604.20183","last_updated":"2026-06-02T10:46:47Z","snapshot_observed_at":"2026-08-02T22:19:27.963593Z","submitted_at":"2026-04-22T04:55:31Z","title":"Dual-Cluster Memory Agent: Resolving Multi-Paradigm Ambiguity in Optimization Problem Solving","version":1},"reference_index":80,"source":"arxiv_source","source_observed_at":"2026-05-10T00:37:55.147350Z"},"links":{"cited_paper":"/paper/2501.16411","citing_paper":"/paper/2604.20183"},"observation_digest":"sha256:2f1a1eb2b9b36c2bbc276b8463a4531f498eb4a2086fc5d5a1efcffd305b4cf6","observation_id":"4b761ef7-61fa-437e-8043-744e943c169f","resolution":{"observed_at":"2026-05-11T13:46:04.347333Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","version":2},"cited_work":{"arxiv_id":"2501.16411","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.16411","snapshot_observed_at":"2026-07-09T18:06:25.898385Z","title":"PhysBench: Benchmarking and enhancing vision-language models for physical world understanding","venue":"cs.CV","work_id":"3a5e1663-11b8-43aa-8e78-33ab23326881","year":2025},"citing_paper":{"arxiv_id":"2604.21510","last_updated":"2026-04-23T10:12:32Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-23T10:12:32Z","title":"OptiVerse: A Comprehensive Benchmark towards Optimization Problem Solving","version":1},"reference_index":96,"source":"arxiv_source","source_observed_at":"2026-05-09T22:04:19.654714Z"},"links":{"cited_paper":"/paper/2501.16411","citing_paper":"/paper/2604.21510"},"observation_digest":"sha256:4d217f0af3fc5b0ff624923f69f8fed8078cf7a49836c1359d60c340ca5f3f3e","observation_id":"c6304c87-1c74-40c7-9399-168e0d4138f5","resolution":{"observed_at":"2026-05-11T14:21:03.894039Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","version":2},"cited_work":{"arxiv_id":"2501.16411","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.16411","snapshot_observed_at":"2026-07-09T18:06:25.898385Z","title":"PhysBench: Benchmarking and enhancing vision-language models for physical world understanding","venue":"cs.CV","work_id":"3a5e1663-11b8-43aa-8e78-33ab23326881","year":2025},"citing_paper":{"arxiv_id":"2604.21873","last_updated":"2026-04-23T17:17:18Z","snapshot_observed_at":"2026-07-06T23:08:23.939574Z","submitted_at":"2026-04-23T17:17:18Z","title":"Grounding Video Reasoning in Physical Signals","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-09T22:29:21.240010Z"},"links":{"cited_paper":"/paper/2501.16411","citing_paper":"/paper/2604.21873"},"observation_digest":"sha256:8f2d820a7c2ab765458531bc6ba1828d5b2c07c8465de5d9ee46af46d143d4ac","observation_id":"935b2493-3073-40e1-bb43-26b7cdfca23a","resolution":{"observed_at":"2026-05-09T22:34:07.272291Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","version":2},"cited_work":{"arxiv_id":"2501.16411","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.16411","snapshot_observed_at":"2026-07-09T18:06:25.898385Z","title":"PhysBench: Benchmarking and enhancing vision-language models for physical world understanding","venue":"cs.CV","work_id":"3a5e1663-11b8-43aa-8e78-33ab23326881","year":2025},"citing_paper":{"arxiv_id":"2605.04515","last_updated":"2026-05-06T05:48:56Z","snapshot_observed_at":"2026-07-06T23:17:18.486586Z","submitted_at":"2026-05-06T05:48:56Z","title":"From Priors to Perception: Grounding Video-LLMs in Physical Reality","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-08T17:41:23.233366Z"},"links":{"cited_paper":"/paper/2501.16411","citing_paper":"/paper/2605.04515"},"observation_digest":"sha256:7823ae83be197ae5981dff509684379ad45e3fda8c35c1913f770cb922f7dbb1","observation_id":"f65b4cce-0ad5-4843-b705-438b4bd60786","resolution":{"observed_at":"2026-05-11T17:21:08.313979Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","version":2},"cited_work":{"arxiv_id":"2501.16411","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.16411","snapshot_observed_at":"2026-07-09T18:06:25.898385Z","title":"PhysBench: Benchmarking and enhancing vision-language models for physical world understanding","venue":"cs.CV","work_id":"3a5e1663-11b8-43aa-8e78-33ab23326881","year":2025},"citing_paper":{"arxiv_id":"2605.07568","last_updated":"2026-05-08T10:40:08Z","snapshot_observed_at":"2026-07-06T23:19:56.415523Z","submitted_at":"2026-05-08T10:40:08Z","title":"Tracing the Arrow of Time: Diagnosing Temporal Information Flow in Video-LLMs","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-11T03:04:27.841522Z"},"links":{"cited_paper":"/paper/2501.16411","citing_paper":"/paper/2605.07568"},"observation_digest":"sha256:5f513b272c0a97c6f087788afaeb026e63829c99bd2562c3d194b896443aba13","observation_id":"4ef3b87a-c965-4d11-b366-5270fa6a8c92","resolution":{"observed_at":"2026-05-11T03:05:53.207696Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","version":2},"cited_work":{"arxiv_id":"2501.16411","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.16411","snapshot_observed_at":"2026-07-09T18:06:25.898385Z","title":"PhysBench: Benchmarking and enhancing vision-language models for physical world understanding","venue":"cs.CV","work_id":"3a5e1663-11b8-43aa-8e78-33ab23326881","year":2025},"citing_paper":{"arxiv_id":"2605.15185","last_updated":"2026-05-14T17:59:04Z","snapshot_observed_at":"2026-08-03T12:57:56.209990Z","submitted_at":"2026-05-14T17:59:04Z","title":"Quantitative Video World Model Evaluation for Geometric-Consistency","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-15T03:11:03.060052Z"},"links":{"cited_paper":"/paper/2501.16411","citing_paper":"/paper/2605.15185"},"observation_digest":"sha256:17f10ce41046ccbc03a2501f68ec776ccfcee482b0f5dffb125007ffe095491d","observation_id":"2d14ffac-878a-4c03-aa2e-43ef42ff1033","resolution":{"observed_at":"2026-05-15T03:14:53.082576Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","version":2},"cited_work":{"arxiv_id":"2501.16411","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.16411","snapshot_observed_at":"2026-07-09T18:06:25.898385Z","title":"PhysBench: Benchmarking and enhancing vision-language models for physical world understanding","venue":"cs.CV","work_id":"3a5e1663-11b8-43aa-8e78-33ab23326881","year":2025},"citing_paper":{"arxiv_id":"2605.15298","last_updated":"2026-05-14T18:11:47Z","snapshot_observed_at":"2026-08-02T19:33:56.173371Z","submitted_at":"2026-05-14T18:11:47Z","title":"PhysBrain 1.0 Technical Report","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-19T16:34:44.204055Z"},"links":{"cited_paper":"/paper/2501.16411","citing_paper":"/paper/2605.15298"},"observation_digest":"sha256:16fdaf8ee02e2c34a40fb4f7505ff2897b84602e77a9007f48a0e0f885f4f2ea","observation_id":"6ab7aa76-0d97-4a16-b022-74df7f7b3ab3","resolution":{"observed_at":"2026-05-19T16:37:39.921164Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","version":2},"cited_work":{"arxiv_id":"2501.16411","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.16411","snapshot_observed_at":"2026-07-09T18:06:25.898385Z","title":"PhysBench: Benchmarking and enhancing vision-language models for physical world understanding","venue":"cs.CV","work_id":"3a5e1663-11b8-43aa-8e78-33ab23326881","year":2025},"citing_paper":{"arxiv_id":"2605.16292","last_updated":"2026-04-14T14:10:12Z","snapshot_observed_at":"2026-07-06T23:27:29.931962Z","submitted_at":"2026-04-14T14:10:12Z","title":"Evidence of a Cognitive Shift in AI Education: How Students Are Rethinking Human Intelligence?","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-21T01:31:17.669186Z"},"links":{"cited_paper":"/paper/2501.16411","citing_paper":"/paper/2605.16292"},"observation_digest":"sha256:77ecc00fd778908fcf31419c3ab41670fd4e9324f227aa2ac9c12e33abd1a573","observation_id":"a84fa5a1-28c9-4685-9564-e6224bcd8179","resolution":{"observed_at":"2026-05-21T01:33:56.534774Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","version":2},"cited_work":{"arxiv_id":"2501.16411","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.16411","snapshot_observed_at":"2026-07-09T18:06:25.898385Z","title":"PhysBench: Benchmarking and enhancing vision-language models for physical world understanding","venue":"cs.CV","work_id":"3a5e1663-11b8-43aa-8e78-33ab23326881","year":2025},"citing_paper":{"arxiv_id":"2605.16713","last_updated":"2026-06-11T14:17:29Z","snapshot_observed_at":"2026-07-06T23:27:48.303599Z","submitted_at":"2026-05-15T23:52:11Z","title":"GeoWorld-VLM: Geometry from World Models for Vision-Language Models","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-20T17:57:00.909897Z"},"links":{"cited_paper":"/paper/2501.16411","citing_paper":"/paper/2605.16713"},"observation_digest":"sha256:080c5ca7b4a6bfc1a9c0250b3034d610352bc0dfec15af5143e1f5a9d7cd2089","observation_id":"150f97be-9eaa-4875-9b79-17369e18171e","resolution":{"observed_at":"2026-05-20T17:58:49.557468Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","version":2},"cited_work":{"arxiv_id":"2501.16411","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.16411","snapshot_observed_at":"2026-07-09T18:06:25.898385Z","title":"PhysBench: Benchmarking and enhancing vision-language models for physical world understanding","venue":"cs.CV","work_id":"3a5e1663-11b8-43aa-8e78-33ab23326881","year":2025},"citing_paper":{"arxiv_id":"2605.16713","last_updated":"2026-06-11T14:17:29Z","snapshot_observed_at":"2026-07-06T23:27:48.303599Z","submitted_at":"2026-05-15T23:52:11Z","title":"GeoWorld-VLM: Geometry from World Models for Vision-Language Models","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-30T19:02:05.125937Z"},"links":{"cited_paper":"/paper/2501.16411","citing_paper":"/paper/2605.16713"},"observation_digest":"sha256:548b8235c4d8fbe189e98e5f79074a89176f386c6163da4db5376b903b0903bd","observation_id":"637f56f8-b75c-46b0-8cd7-9eef946ff3c5","resolution":{"observed_at":"2026-06-30T19:05:00.630085Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","version":2},"cited_work":{"arxiv_id":"2501.16411","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.16411","snapshot_observed_at":"2026-07-09T18:06:25.898385Z","title":"PhysBench: Benchmarking and enhancing vision-language models for physical world understanding","venue":"cs.CV","work_id":"3a5e1663-11b8-43aa-8e78-33ab23326881","year":2025},"citing_paper":{"arxiv_id":"2605.18746","last_updated":"2026-05-25T08:34:52Z","snapshot_observed_at":"2026-08-02T07:58:31.278661Z","submitted_at":"2026-05-18T17:59:02Z","title":"ESI-Bench: Towards Embodied Spatial Intelligence that Closes the Perception-Action Loop","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-20T10:52:22.778489Z"},"links":{"cited_paper":"/paper/2501.16411","citing_paper":"/paper/2605.18746"},"observation_digest":"sha256:8b97eea6ed5b3e37aab5aba615185a2f3aa27b7a343312f70b4ffbbd89e116ba","observation_id":"6fdda63d-6d25-48c9-a74e-4b9c0c25729a","resolution":{"observed_at":"2026-05-20T10:53:13.228601Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","version":2},"cited_work":{"arxiv_id":"2501.16411","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.16411","snapshot_observed_at":"2026-07-09T18:06:25.898385Z","title":"PhysBench: Benchmarking and enhancing vision-language models for physical world understanding","venue":"cs.CV","work_id":"3a5e1663-11b8-43aa-8e78-33ab23326881","year":2025},"citing_paper":{"arxiv_id":"2605.18746","last_updated":"2026-05-25T08:34:52Z","snapshot_observed_at":"2026-08-02T07:58:31.278661Z","submitted_at":"2026-05-18T17:59:02Z","title":"ESI-Bench: Towards Embodied Spatial Intelligence that Closes the Perception-Action Loop","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-30T18:25:17.831116Z"},"links":{"cited_paper":"/paper/2501.16411","citing_paper":"/paper/2605.18746"},"observation_digest":"sha256:192d1c13414a7fd1b9db3e8395bc2fafa694df0d4c8e36fd43d84fc6a0c510dc","observation_id":"20dda940-1c8b-48fd-9f8f-0d7591ca878a","resolution":{"observed_at":"2026-07-01T14:55:48.810709Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","version":2},"cited_work":{"arxiv_id":"2501.16411","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.16411","snapshot_observed_at":"2026-07-09T18:06:25.898385Z","title":"PhysBench: Benchmarking and enhancing vision-language models for physical world understanding","venue":"cs.CV","work_id":"3a5e1663-11b8-43aa-8e78-33ab23326881","year":2025},"citing_paper":{"arxiv_id":"2605.20576","last_updated":"2026-05-20T00:23:56Z","snapshot_observed_at":"2026-08-02T21:02:37.103727Z","submitted_at":"2026-05-20T00:23:56Z","title":"$\\Delta$ynamics: Language-Based Representation for Inferring Rigid-Body Dynamics From Videos","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-21T06:14:26.824489Z"},"links":{"cited_paper":"/paper/2501.16411","citing_paper":"/paper/2605.20576"},"observation_digest":"sha256:f1628387e4591f2974bb51229415d498e36bc96a58c7a4a02a689ea097445d1e","observation_id":"e87495d2-7025-455f-9942-efeb2ac24e45","resolution":{"observed_at":"2026-05-21T06:14:40.989000Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","version":2},"cited_work":{"arxiv_id":"2501.16411","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.16411","snapshot_observed_at":"2026-07-09T18:06:25.898385Z","title":"PhysBench: Benchmarking and enhancing vision-language models for physical world understanding","venue":"cs.CV","work_id":"3a5e1663-11b8-43aa-8e78-33ab23326881","year":2025},"citing_paper":{"arxiv_id":"2605.29585","last_updated":"2026-05-28T08:29:32Z","snapshot_observed_at":"2026-08-03T01:46:36.319850Z","submitted_at":"2026-05-28T08:29:32Z","title":"World Models in Words: Auditing Physical State-Transition Commitments in Vision-Language Models","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-29T08:18:45.804447Z"},"links":{"cited_paper":"/paper/2501.16411","citing_paper":"/paper/2605.29585"},"observation_digest":"sha256:702ac4f6528f4321faa09e2f75574e654f3cc5cc8e3cdf5833022bc826ee602f","observation_id":"5ead1bbb-04be-4290-af55-1656208acec8","resolution":{"observed_at":"2026-06-29T08:23:15.377444Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","version":2},"cited_work":{"arxiv_id":"2501.16411","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.16411","snapshot_observed_at":"2026-07-09T18:06:25.898385Z","title":"PhysBench: Benchmarking and enhancing vision-language models for physical world understanding","venue":"cs.CV","work_id":"3a5e1663-11b8-43aa-8e78-33ab23326881","year":2025},"citing_paper":{"arxiv_id":"2605.30339","last_updated":"2026-05-28T17:59:09Z","snapshot_observed_at":"2026-08-02T20:03:38.580074Z","submitted_at":"2026-05-28T17:59:09Z","title":"Benchmarking Single-Factor Physical Video-to-Audio Generation","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-29T07:41:56.917119Z"},"links":{"cited_paper":"/paper/2501.16411","citing_paper":"/paper/2605.30339"},"observation_digest":"sha256:23a61eaf337c9a159628c9adc53249be9ebbeb28be7989b461096fc4433e7030","observation_id":"6b6d6710-75cd-4995-8649-3336d863e737","resolution":{"observed_at":"2026-06-29T07:43:13.382683Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","version":2},"cited_work":{"arxiv_id":"2501.16411","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.16411","snapshot_observed_at":"2026-07-09T18:06:25.898385Z","title":"PhysBench: Benchmarking and enhancing vision-language models for physical world understanding","venue":"cs.CV","work_id":"3a5e1663-11b8-43aa-8e78-33ab23326881","year":2025},"citing_paper":{"arxiv_id":"2605.30542","last_updated":"2026-05-28T20:18:22Z","snapshot_observed_at":"2026-07-06T23:39:48.531480Z","submitted_at":"2026-05-28T20:18:22Z","title":"Physically Viable World Models: A Case for Query-Conditioned Embodied AI","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-29T06:55:57.801162Z"},"links":{"cited_paper":"/paper/2501.16411","citing_paper":"/paper/2605.30542"},"observation_digest":"sha256:cae5a54fbd806d920e8fd911d95f0ba3b43b8316ce8b8d7739ed136cc57521fc","observation_id":"dbf71278-e1e7-40ad-8abb-55308a286db9","resolution":{"observed_at":"2026-06-29T09:13:16.514166Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","version":2},"cited_work":{"arxiv_id":"2501.16411","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.16411","snapshot_observed_at":"2026-07-09T18:06:25.898385Z","title":"PhysBench: Benchmarking and enhancing vision-language models for physical world understanding","venue":"cs.CV","work_id":"3a5e1663-11b8-43aa-8e78-33ab23326881","year":2025},"citing_paper":{"arxiv_id":"2606.05966","last_updated":"2026-06-04T10:07:05Z","snapshot_observed_at":"2026-07-06T23:45:51.910263Z","submitted_at":"2026-06-04T10:07:05Z","title":"Causal Scaffolding for Physical Reasoning: A Benchmark for Causally-Informed Physical World Understanding in VLMs","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-27T23:15:38.962013Z"},"links":{"cited_paper":"/paper/2501.16411","citing_paper":"/paper/2606.05966"},"observation_digest":"sha256:d52ef7b4b03f84b0d587b009aef968d5020368cdda9b48cd75ebfd254446e2a3","observation_id":"99ed44cd-9b49-4b80-b54e-42c1ffa78105","resolution":{"observed_at":"2026-07-02T15:57:06.745261Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","version":2},"cited_work":{"arxiv_id":"2501.16411","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.16411","snapshot_observed_at":"2026-07-09T18:06:25.898385Z","title":"PhysBench: Benchmarking and enhancing vision-language models for physical world understanding","venue":"cs.CV","work_id":"3a5e1663-11b8-43aa-8e78-33ab23326881","year":2025},"citing_paper":{"arxiv_id":"2606.06361","last_updated":"2026-06-17T07:38:02Z","snapshot_observed_at":"2026-08-03T04:40:21.011532Z","submitted_at":"2026-06-04T16:30:39Z","title":"Physics in 2-Steps: Locking Motion Priors Before Visual Refinement Erases Them","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-06-28T01:58:00.336605Z"},"links":{"cited_paper":"/paper/2501.16411","citing_paper":"/paper/2606.06361"},"observation_digest":"sha256:f5daec293a6f61184a737e5861ebefad605245b0a8ce57b0d61849a635427f1b","observation_id":"03672721-5de8-42b2-84a0-91793d035cc0","resolution":{"observed_at":"2026-07-02T12:36:57.143795Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","version":2},"cited_work":{"arxiv_id":"2501.16411","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.16411","snapshot_observed_at":"2026-07-09T18:06:25.898385Z","title":"PhysBench: Benchmarking and enhancing vision-language models for physical world understanding","venue":"cs.CV","work_id":"3a5e1663-11b8-43aa-8e78-33ab23326881","year":2025},"citing_paper":{"arxiv_id":"2606.07962","last_updated":"2026-06-06T03:40:47Z","snapshot_observed_at":"2026-07-31T20:27:39.414307Z","submitted_at":"2026-06-06T03:40:47Z","title":"ChronoPhyBench: Do MLLMs Truly Understand the World or Merely Exploit Language Priors?","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-06-27T20:23:18.667677Z"},"links":{"cited_paper":"/paper/2501.16411","citing_paper":"/paper/2606.07962"},"observation_digest":"sha256:739882d5c9e5f00518800fd9c7b703b3e516964272d18f73819635fef5ab4682","observation_id":"d8f0afaf-4af6-4b32-b691-5135409340bd","resolution":{"observed_at":"2026-07-02T20:27:22.649811Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","version":2},"cited_work":{"arxiv_id":"2501.16411","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.16411","snapshot_observed_at":"2026-07-09T18:06:25.898385Z","title":"PhysBench: Benchmarking and enhancing vision-language models for physical world understanding","venue":"cs.CV","work_id":"3a5e1663-11b8-43aa-8e78-33ab23326881","year":2025},"citing_paper":{"arxiv_id":"2606.25212","last_updated":"2026-06-25T06:30:32Z","snapshot_observed_at":"2026-08-04T22:37:37.317283Z","submitted_at":"2026-06-23T22:15:42Z","title":"RigPI: Dynamic Parameter Identification of Rigid Body via VLM-Seeded Differentiable Simulation","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-25T23:35:13.949488Z"},"links":{"cited_paper":"/paper/2501.16411","citing_paper":"/paper/2606.25212"},"observation_digest":"sha256:8eafd9f4d13af3163a7c8dcb47959f9611982fad4936cdfc318e0a6136de30a3","observation_id":"75b7dd1d-4257-4bb6-9d38-070252d0be41","resolution":{"observed_at":"2026-07-04T17:39:59.976044Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","version":2},"cited_work":{"arxiv_id":"2501.16411","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.16411","snapshot_observed_at":"2026-07-09T18:06:25.898385Z","title":"PhysBench: Benchmarking and enhancing vision-language models for physical world understanding","venue":"cs.CV","work_id":"3a5e1663-11b8-43aa-8e78-33ab23326881","year":2025},"citing_paper":{"arxiv_id":"2606.25212","last_updated":"2026-06-25T06:30:32Z","snapshot_observed_at":"2026-08-04T22:37:37.317283Z","submitted_at":"2026-06-23T22:15:42Z","title":"RigPI: Dynamic Parameter Identification of Rigid Body via VLM-Seeded Differentiable Simulation","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-26T05:22:09.771975Z"},"links":{"cited_paper":"/paper/2501.16411","citing_paper":"/paper/2606.25212"},"observation_digest":"sha256:468347c9ed98c866dcfccd37b9e26f6b01367de0d9723c3bc2d0cc6ae5259cc6","observation_id":"8fac81bd-607e-4c0c-a713-b8d29f4845d8","resolution":{"observed_at":"2026-07-04T13:19:50.512556Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","version":2},"cited_work":{"arxiv_id":"2501.16411","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.16411","snapshot_observed_at":"2026-07-09T18:06:25.898385Z","title":"PhysBench: Benchmarking and enhancing vision-language models for physical world understanding","venue":"cs.CV","work_id":"3a5e1663-11b8-43aa-8e78-33ab23326881","year":2025},"citing_paper":{"arxiv_id":"2607.00881","last_updated":"2026-07-01T12:45:12Z","snapshot_observed_at":"2026-08-03T01:24:10.985411Z","submitted_at":"2026-07-01T12:45:12Z","title":"OmniView-Space: Reinforcing Spatial Reasoning via Multi-Perspective Spatial Mapping","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-07-02T14:16:39.649823Z"},"links":{"cited_paper":"/paper/2501.16411","citing_paper":"/paper/2607.00881"},"observation_digest":"sha256:47cfbe3eee0ba0172e77624d6f1de4ddc1987b3b81a5e7392b380a2e62e839b0","observation_id":"3de5026f-5932-46d8-afa2-1f54c803ea22","resolution":{"observed_at":"2026-07-02T14:17:02.469095Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","version":2},"cited_work":{"arxiv_id":"2501.16411","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.16411","snapshot_observed_at":"2026-07-09T18:06:25.898385Z","title":"PhysBench: Benchmarking and enhancing vision-language models for physical world understanding","venue":"cs.CV","work_id":"3a5e1663-11b8-43aa-8e78-33ab23326881","year":2025},"citing_paper":{"arxiv_id":"2607.07189","last_updated":"2026-07-08T09:23:56Z","snapshot_observed_at":"2026-08-04T06:44:45.483597Z","submitted_at":"2026-07-08T09:23:56Z","title":"Does AI Understand Imaging? A Systematic Benchmark of Agentic AI for Computational Imaging Tasks","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-09T17:59:03.869202Z"},"links":{"cited_paper":"/paper/2501.16411","citing_paper":"/paper/2607.07189"},"observation_digest":"sha256:66a059361b1d8bdf29669a5bf5355632bb4e76f0e6e5c59347a97f1e5cdf8e6d","observation_id":"e04f13be-3fad-462a-862b-e35eaebd4cb3","resolution":{"observed_at":"2026-07-09T18:06:25.899637Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.16411","snapshot_observed_at":"2026-07-14T07:27:08.281893Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.11044","last_updated":"2026-07-13T03:13:39Z","snapshot_observed_at":"2026-08-03T15:44:46.398917Z","submitted_at":"2026-07-13T03:13:39Z","title":"RetroHolmes: When Semantic Plausibility Fails Retrospective Physical Process Reasoning","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-07-14T07:27:08.281893Z"},"links":{"cited_paper":"/paper/2501.16411","citing_paper":"/paper/2607.11044"},"observation_digest":"sha256:4cfce12fa8802a29b6b2a61df87e4e9733a5a69c62a44898f7fd5b32f8aedadb","observation_id":"8299f42c-fbf6-4c4a-bca0-c50e31843b7c","resolution":{"observed_at":"2026-07-14T07:27:08.281893Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.16411","snapshot_observed_at":"2026-08-01T21:07:24.575800Z","title":"Physbench: Benchmarking and enhancing vision-language models for physical world understanding, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16401","last_updated":"2026-07-17T18:00:05Z","snapshot_observed_at":"2026-08-02T23:36:27.285941Z","submitted_at":"2026-07-17T18:00:05Z","title":"Apple-$\\pi$: Benchmarking Thinking with Video Towards Law-Grounded Physical Intelligence","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-01T21:07:24.575800Z"},"links":{"cited_paper":"/paper/2501.16411","citing_paper":"/paper/2607.16401"},"observation_digest":"sha256:2981850ee70416886e9dcdb1d926f046f4e60eef86119d73ae79a6aab82afef2","observation_id":"1c6b2d5e-c3fd-4ab7-922d-cf3aabae0e3b","resolution":{"observed_at":"2026-08-01T21:07:24.575800Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2501.16411/citation-record","integrity":"/paper/2501.16411/integrity","json":"/paper/2501.16411/citation-record.json","paper":"/paper/2501.16411"},"outbound":[],"paper":{"arxiv_id":"2501.16411","last_updated":"2025-01-29T03:52:39Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-02T23:35:21.038166Z","submitted_at":"2025-01-27T18:59:58Z","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 5 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 33 inbound Pith citation observations for arXiv:2501.16411."}