{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:7LNV4V7UQONARXQOD3R3QMSPS5","short_pith_number":"pith:7LNV4V7U","schema_version":"1.0","canonical_sha256":"fadb5e57f4839a08de0e1ee3b8324f97705273a566a941d20cc988764a278ac8","source":{"kind":"arxiv","id":"2509.02359","version":1},"attestation_state":"computed","paper":{"title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Helu Zhi, Jiajun Zhang, Jingjing Huang, Shuo Ren, Wang Xu, Wanyue Zhang, Yangbin Xu, Yibin Huang","submitted_at":"2025-09-02T14:22:43Z","abstract_excerpt":"Spatial understanding is essential for Multimodal Large Language Models (MLLMs) to support perception, reasoning, and planning in embodied environments. Despite recent progress, existing studies reveal that MLLMs still struggle with spatial understanding. However, existing research lacks a comprehensive and systematic evaluation of these limitations, often restricted to isolated scenarios, such as single-view or video. In this work, we present a systematic analysis of spatial understanding from both data and architectural perspectives across three representative scenarios: single-view, multi-v"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.02359","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-09-02T14:22:43Z","cross_cats_sorted":[],"title_canon_sha256":"5d7a7f2c79b0ab49d583ca5fd227249dc25cd9a1225873b8ede3c7e28851a9fe","abstract_canon_sha256":"7f98d856c70ab3c6eac08edd50af1022dda7506160d5cfedf3a0c3f66a0d3b2b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:03:37.072469Z","signature_b64":"kvYNpsA4Rb/MRe3TnzFjGDCeK3PaZ3Il23toSyYACOrYANdQgIxRs7kuIZlAgs1U59xxfv36gYJumXYdfKUZCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fadb5e57f4839a08de0e1ee3b8324f97705273a566a941d20cc988764a278ac8","last_reissued_at":"2026-07-05T12:03:37.071820Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:03:37.071820Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Helu Zhi, Jiajun Zhang, Jingjing Huang, Shuo Ren, Wang Xu, Wanyue Zhang, Yangbin Xu, Yibin Huang","submitted_at":"2025-09-02T14:22:43Z","abstract_excerpt":"Spatial understanding is essential for Multimodal Large Language Models (MLLMs) to support perception, reasoning, and planning in embodied environments. Despite recent progress, existing studies reveal that MLLMs still struggle with spatial understanding. However, existing research lacks a comprehensive and systematic evaluation of these limitations, often restricted to isolated scenarios, such as single-view or video. In this work, we present a systematic analysis of spatial understanding from both data and architectural perspectives across three representative scenarios: single-view, multi-v"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.02359","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.02359/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.02359","created_at":"2026-07-05T12:03:37.071897+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.02359v1","created_at":"2026-07-05T12:03:37.071897+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.02359","created_at":"2026-07-05T12:03:37.071897+00:00"},{"alias_kind":"pith_short_12","alias_value":"7LNV4V7UQONA","created_at":"2026-07-05T12:03:37.071897+00:00"},{"alias_kind":"pith_short_16","alias_value":"7LNV4V7UQONARXQO","created_at":"2026-07-05T12:03:37.071897+00:00"},{"alias_kind":"pith_short_8","alias_value":"7LNV4V7U","created_at":"2026-07-05T12:03:37.071897+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25634","citing_title":"SSMNBench: Diagnosing Image-based Cross-View Human-Object Understanding via Single-View Sufficiency and Multi-View Necessity","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11770","citing_title":"SVoT: State-aware Visualization-of-Thought for Spatial Reasoning via Reinforcement Learning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22100","citing_title":"MPDocBench-Parse: Benchmarking Practical Multi-page Document Parsing","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30161","citing_title":"Why Far Looks Up: Probing Spatial Representation in Vision-Language Models","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22100","citing_title":"MPDocBench-Parse: Benchmarking Practical Multi-page Document Parsing","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2603.03944","citing_title":"SCP: Spatial Causal Prediction in Video","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13321","citing_title":"Why MLLMs Struggle to Determine Object Orientations","ref_index":42,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7LNV4V7UQONARXQOD3R3QMSPS5","json":"https://pith.science/pith/7LNV4V7UQONARXQOD3R3QMSPS5.json","graph_json":"https://pith.science/api/pith-number/7LNV4V7UQONARXQOD3R3QMSPS5/graph.json","events_json":"https://pith.science/api/pith-number/7LNV4V7UQONARXQOD3R3QMSPS5/events.json","paper":"https://pith.science/paper/7LNV4V7U"},"agent_actions":{"view_html":"https://pith.science/pith/7LNV4V7UQONARXQOD3R3QMSPS5","download_json":"https://pith.science/pith/7LNV4V7UQONARXQOD3R3QMSPS5.json","view_paper":"https://pith.science/paper/7LNV4V7U","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.02359&json=true","fetch_graph":"https://pith.science/api/pith-number/7LNV4V7UQONARXQOD3R3QMSPS5/graph.json","fetch_events":"https://pith.science/api/pith-number/7LNV4V7UQONARXQOD3R3QMSPS5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7LNV4V7UQONARXQOD3R3QMSPS5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7LNV4V7UQONARXQOD3R3QMSPS5/action/storage_attestation","attest_author":"https://pith.science/pith/7LNV4V7UQONARXQOD3R3QMSPS5/action/author_attestation","sign_citation":"https://pith.science/pith/7LNV4V7UQONARXQOD3R3QMSPS5/action/citation_signature","submit_replication":"https://pith.science/pith/7LNV4V7UQONARXQOD3R3QMSPS5/action/replication_record"}},"created_at":"2026-07-05T12:03:37.071897+00:00","updated_at":"2026-07-05T12:03:37.071897+00:00"}