{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:L5JB2337VTSDJRXCAHRFKXFDFH","short_pith_number":"pith:L5JB2337","schema_version":"1.0","canonical_sha256":"5f521d6f7face434c6e201e2555ca329e50ee5decdae2d1b216c9c6eaabb8bd1","source":{"kind":"arxiv","id":"2501.16411","version":2},"attestation_state":"computed","paper":{"title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG","cs.RO"],"primary_cat":"cs.CV","authors_text":"Boyi Li, Daniel Seita, Jiageng Mao, Vitor Guizilini, Wei Chow, Yue Wang","submitted_at":"2025-01-27T18:59:58Z","abstract_excerpt":"Understanding the physical world is a fundamental challenge in embodied AI, critical for enabling agents to perform complex tasks and operate safely in real-world environments. While Vision-Language Models (VLMs) have shown great promise in reasoning and task planning for embodied agents, their ability to comprehend physical phenomena remains extremely limited. To close this gap, we introduce PhysBench, a comprehensive benchmark designed to evaluate VLMs' physical world understanding capability across a diverse set of tasks. PhysBench contains 10,002 entries of interleaved video-image-text dat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.16411","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-01-27T18:59:58Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG","cs.RO"],"title_canon_sha256":"84bee8c6391382b0578069050500cc4db7b1784705f6aa47aaf43166c88f1039","abstract_canon_sha256":"f939fbeb41379bbcb261c1688c72fd36c80537890cd24b8510a123505f176e5a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:06:47.959846Z","signature_b64":"aZ+woXldk/Bjnzvb13nBBJmeaHoJeePDPltk+CxKzXs8Etx+E/h4JeKmZqLNAYwmy4RWiWd/J2LZzyAVEVwNBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5f521d6f7face434c6e201e2555ca329e50ee5decdae2d1b216c9c6eaabb8bd1","last_reissued_at":"2026-07-05T10:06:47.959353Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:06:47.959353Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG","cs.RO"],"primary_cat":"cs.CV","authors_text":"Boyi Li, Daniel Seita, Jiageng Mao, Vitor Guizilini, Wei Chow, Yue Wang","submitted_at":"2025-01-27T18:59:58Z","abstract_excerpt":"Understanding the physical world is a fundamental challenge in embodied AI, critical for enabling agents to perform complex tasks and operate safely in real-world environments. While Vision-Language Models (VLMs) have shown great promise in reasoning and task planning for embodied agents, their ability to comprehend physical phenomena remains extremely limited. To close this gap, we introduce PhysBench, a comprehensive benchmark designed to evaluate VLMs' physical world understanding capability across a diverse set of tasks. PhysBench contains 10,002 entries of interleaved video-image-text dat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.16411","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.16411/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.16411","created_at":"2026-07-05T10:06:47.959415+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.16411v2","created_at":"2026-07-05T10:06:47.959415+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.16411","created_at":"2026-07-05T10:06:47.959415+00:00"},{"alias_kind":"pith_short_12","alias_value":"L5JB2337VTSD","created_at":"2026-07-05T10:06:47.959415+00:00"},{"alias_kind":"pith_short_16","alias_value":"L5JB2337VTSDJRXC","created_at":"2026-07-05T10:06:47.959415+00:00"},{"alias_kind":"pith_short_8","alias_value":"L5JB2337","created_at":"2026-07-05T10:06:47.959415+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":28,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07189","citing_title":"Does AI Understand Imaging? A Systematic Benchmark of Agentic AI for Computational Imaging Tasks","ref_index":10,"is_internal_anchor":true},{"citing_arxiv_id":"2606.25212","citing_title":"RigPI: Dynamic Parameter Identification of Rigid Body via VLM-Seeded Differentiable Simulation","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25212","citing_title":"RigPI: Dynamic Parameter Identification of Rigid Body via VLM-Seeded Differentiable Simulation","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07962","citing_title":"ChronoPhyBench: Do MLLMs Truly Understand the World or Merely Exploit Language Priors?","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05966","citing_title":"Causal Scaffolding for Physical Reasoning: A Benchmark for Causally-Informed Physical World Understanding in VLMs","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00881","citing_title":"OmniView-Space: Reinforcing Spatial Reasoning via Multi-Perspective Spatial Mapping","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06361","citing_title":"Physics in 2-Steps: Locking Motion Priors Before Visual Refinement Erases Them","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18746","citing_title":"ESI-Bench: Towards Embodied Spatial Intelligence that Closes the Perception-Action Loop","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16713","citing_title":"GeoWorld-VLM: Geometry from World Models for Vision-Language Models","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30542","citing_title":"Physically Viable World Models: A Case for Query-Conditioned Embodied AI","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29585","citing_title":"World Models in Words: Auditing Physical State-Transition Commitments in Vision-Language Models","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30339","citing_title":"Benchmarking Single-Factor Physical Video-to-Audio Generation","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2512.23292","citing_title":"Agentic Physical AI toward a Domain-Specific Foundation Model for Nuclear Reactor Control","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20576","citing_title":"$\\Delta$ynamics: Language-Based Representation for Inferring Rigid-Body Dynamics From Videos","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16292","citing_title":"Evidence of a Cognitive Shift in AI Education: How Students Are Rethinking Human Intelligence?","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16713","citing_title":"GeoWorld-VLM: Geometry from World Models for Vision-Language Models","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18746","citing_title":"ESI-Bench: Towards Embodied Spatial Intelligence that Closes the Perception-Action Loop","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15298","citing_title":"PhysBrain 1.0 Technical Report","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2511.10946","citing_title":"Abstract 3D Perception for Spatial Intelligence in Vision-Language Models","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2511.18373","citing_title":"MASS: Motion-Aware Spatial-Temporal Grounding for Physics Reasoning and Comprehension in Vision-Language Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2603.03944","citing_title":"SCP: Spatial Causal Prediction in Video","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15185","citing_title":"Quantitative Video World Model Evaluation for Geometric-Consistency","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.00799","citing_title":"Multimodal Language Models Cannot Spot Spatial Inconsistencies","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04515","citing_title":"From Priors to Perception: Grounding Video-LLMs in Physical Reality","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21510","citing_title":"OptiVerse: A Comprehensive Benchmark towards Optimization Problem Solving","ref_index":96,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/L5JB2337VTSDJRXCAHRFKXFDFH","json":"https://pith.science/pith/L5JB2337VTSDJRXCAHRFKXFDFH.json","graph_json":"https://pith.science/api/pith-number/L5JB2337VTSDJRXCAHRFKXFDFH/graph.json","events_json":"https://pith.science/api/pith-number/L5JB2337VTSDJRXCAHRFKXFDFH/events.json","paper":"https://pith.science/paper/L5JB2337"},"agent_actions":{"view_html":"https://pith.science/pith/L5JB2337VTSDJRXCAHRFKXFDFH","download_json":"https://pith.science/pith/L5JB2337VTSDJRXCAHRFKXFDFH.json","view_paper":"https://pith.science/paper/L5JB2337","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.16411&json=true","fetch_graph":"https://pith.science/api/pith-number/L5JB2337VTSDJRXCAHRFKXFDFH/graph.json","fetch_events":"https://pith.science/api/pith-number/L5JB2337VTSDJRXCAHRFKXFDFH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/L5JB2337VTSDJRXCAHRFKXFDFH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/L5JB2337VTSDJRXCAHRFKXFDFH/action/storage_attestation","attest_author":"https://pith.science/pith/L5JB2337VTSDJRXCAHRFKXFDFH/action/author_attestation","sign_citation":"https://pith.science/pith/L5JB2337VTSDJRXCAHRFKXFDFH/action/citation_signature","submit_replication":"https://pith.science/pith/L5JB2337VTSDJRXCAHRFKXFDFH/action/replication_record"}},"created_at":"2026-07-05T10:06:47.959415+00:00","updated_at":"2026-07-05T10:06:47.959415+00:00"}