{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:2DY2LRBDZJGSS76ZYGZSE5IVIZ","short_pith_number":"pith:2DY2LRBD","schema_version":"1.0","canonical_sha256":"d0f1a5c423ca4d297fd9c1b32275154655e0cf4643b00597a2ba2021413f7eec","source":{"kind":"arxiv","id":"2508.02095","version":2},"attestation_state":"computed","paper":{"title":"VLM4D: Towards Spatiotemporal Awareness in Vision Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Achuta Kadambi, Aditya Nagachandra, Alexander Vilesov, Di Chang, Dongdong Chen, Shijie Zhou, Shuwang Zhang, Xin Eric Wang, Xuehai He, Ziyu Wan","submitted_at":"2025-08-04T06:06:06Z","abstract_excerpt":"Vision language models (VLMs) have shown remarkable capabilities in integrating linguistic and visual reasoning but remain fundamentally limited in understanding dynamic spatiotemporal interactions. Humans effortlessly track and reason about object movements, rotations, and perspective shifts-abilities essential for robust dynamic real-world understanding yet notably lacking in current VLMs. In this paper, we introduce VLM4D, the first benchmark specifically designed to evaluate the spatiotemporal reasoning capabilities of VLMs. Our benchmark comprises diverse real-world and synthetic videos a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.02095","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-08-04T06:06:06Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"d4f59297ad760efb83ac3a5a7c731d2e7bf661e4fb9c2159eb79160ca03666a2","abstract_canon_sha256":"7daa2eb24cbd081f14dfeda74422bf99c24ef6c75914527a0f7ac68fbebbdc76"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:49:59.075106Z","signature_b64":"VGDDryDgHFwKCcpV5TqYd7TJ9O83p/LgfdcxUDX6QHIQ8AQ8ER5797rf7jITlszXqjs4tl8bnLB3u5X0TqJCDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d0f1a5c423ca4d297fd9c1b32275154655e0cf4643b00597a2ba2021413f7eec","last_reissued_at":"2026-07-05T11:49:59.074649Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:49:59.074649Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VLM4D: Towards Spatiotemporal Awareness in Vision Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Achuta Kadambi, Aditya Nagachandra, Alexander Vilesov, Di Chang, Dongdong Chen, Shijie Zhou, Shuwang Zhang, Xin Eric Wang, Xuehai He, Ziyu Wan","submitted_at":"2025-08-04T06:06:06Z","abstract_excerpt":"Vision language models (VLMs) have shown remarkable capabilities in integrating linguistic and visual reasoning but remain fundamentally limited in understanding dynamic spatiotemporal interactions. Humans effortlessly track and reason about object movements, rotations, and perspective shifts-abilities essential for robust dynamic real-world understanding yet notably lacking in current VLMs. In this paper, we introduce VLM4D, the first benchmark specifically designed to evaluate the spatiotemporal reasoning capabilities of VLMs. Our benchmark comprises diverse real-world and synthetic videos a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.02095","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.02095/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.02095","created_at":"2026-07-05T11:49:59.074710+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.02095v2","created_at":"2026-07-05T11:49:59.074710+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.02095","created_at":"2026-07-05T11:49:59.074710+00:00"},{"alias_kind":"pith_short_12","alias_value":"2DY2LRBDZJGS","created_at":"2026-07-05T11:49:59.074710+00:00"},{"alias_kind":"pith_short_16","alias_value":"2DY2LRBDZJGSS76Z","created_at":"2026-07-05T11:49:59.074710+00:00"},{"alias_kind":"pith_short_8","alias_value":"2DY2LRBD","created_at":"2026-07-05T11:49:59.074710+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.23045","citing_title":"The TIME Machine: On The Power of Motion for Efficient Perception","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2511.21471","citing_title":"SpatialBench: Benchmarking Multimodal Large Language Models for Spatial Cognition","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2511.00062","citing_title":"World Simulation with Video Foundation Models for Physical AI","ref_index":100,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2DY2LRBDZJGSS76ZYGZSE5IVIZ","json":"https://pith.science/pith/2DY2LRBDZJGSS76ZYGZSE5IVIZ.json","graph_json":"https://pith.science/api/pith-number/2DY2LRBDZJGSS76ZYGZSE5IVIZ/graph.json","events_json":"https://pith.science/api/pith-number/2DY2LRBDZJGSS76ZYGZSE5IVIZ/events.json","paper":"https://pith.science/paper/2DY2LRBD"},"agent_actions":{"view_html":"https://pith.science/pith/2DY2LRBDZJGSS76ZYGZSE5IVIZ","download_json":"https://pith.science/pith/2DY2LRBDZJGSS76ZYGZSE5IVIZ.json","view_paper":"https://pith.science/paper/2DY2LRBD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.02095&json=true","fetch_graph":"https://pith.science/api/pith-number/2DY2LRBDZJGSS76ZYGZSE5IVIZ/graph.json","fetch_events":"https://pith.science/api/pith-number/2DY2LRBDZJGSS76ZYGZSE5IVIZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2DY2LRBDZJGSS76ZYGZSE5IVIZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2DY2LRBDZJGSS76ZYGZSE5IVIZ/action/storage_attestation","attest_author":"https://pith.science/pith/2DY2LRBDZJGSS76ZYGZSE5IVIZ/action/author_attestation","sign_citation":"https://pith.science/pith/2DY2LRBDZJGSS76ZYGZSE5IVIZ/action/citation_signature","submit_replication":"https://pith.science/pith/2DY2LRBDZJGSS76ZYGZSE5IVIZ/action/replication_record"}},"created_at":"2026-07-05T11:49:59.074710+00:00","updated_at":"2026-07-05T11:49:59.074710+00:00"}