{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GQ4LX5SKKYOQHF2BFT62VYUCCP","short_pith_number":"pith:GQ4LX5SK","schema_version":"1.0","canonical_sha256":"3438bbf64a561d0397412cfdaae28213ddbf7f6938914476e6ae2f15708869cc","source":{"kind":"arxiv","id":"2411.06048","version":1},"attestation_state":"computed","paper":{"title":"An Empirical Analysis on Spatial Reasoning Capabilities of Large Multimodal Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Fatemeh Shiri, Gholamreza Haffari, Mona Golestan Far, Xiao-Yu Guo, Xin Yu, Yuan-Fang Li","submitted_at":"2024-11-09T03:07:33Z","abstract_excerpt":"Large Multimodal Models (LMMs) have achieved strong performance across a range of vision and language tasks. However, their spatial reasoning capabilities are under-investigated. In this paper, we construct a novel VQA dataset, Spatial-MM, to comprehensively study LMMs' spatial understanding and reasoning capabilities. Our analyses on object-relationship and multi-hop reasoning reveal several important findings. Firstly, bounding boxes and scene graphs, even synthetic ones, can significantly enhance LMMs' spatial reasoning. Secondly, LMMs struggle more with questions posed from the human persp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.06048","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-11-09T03:07:33Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"b52b76a0f177b9f6411afbc2fb608ba55ee9cb5fec1efd0c36e70c4926867fe5","abstract_canon_sha256":"63e0ef1e3b19c2e8b3b026322c9033597d656981da003f8de2263e747bcd3eb0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:33:15.640978Z","signature_b64":"Ai0i1AkBj7wnnZNeJ3rF1grTKz7lyqppCbM6RcihhtC6SSBQCN5MpIPCvAUfF3HJN4wEm8jeJ1OmwGivbQscDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3438bbf64a561d0397412cfdaae28213ddbf7f6938914476e6ae2f15708869cc","last_reissued_at":"2026-07-05T09:33:15.640310Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:33:15.640310Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An Empirical Analysis on Spatial Reasoning Capabilities of Large Multimodal Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Fatemeh Shiri, Gholamreza Haffari, Mona Golestan Far, Xiao-Yu Guo, Xin Yu, Yuan-Fang Li","submitted_at":"2024-11-09T03:07:33Z","abstract_excerpt":"Large Multimodal Models (LMMs) have achieved strong performance across a range of vision and language tasks. However, their spatial reasoning capabilities are under-investigated. In this paper, we construct a novel VQA dataset, Spatial-MM, to comprehensively study LMMs' spatial understanding and reasoning capabilities. Our analyses on object-relationship and multi-hop reasoning reveal several important findings. Firstly, bounding boxes and scene graphs, even synthetic ones, can significantly enhance LMMs' spatial reasoning. Secondly, LMMs struggle more with questions posed from the human persp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.06048","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.06048/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.06048","created_at":"2026-07-05T09:33:15.640409+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.06048v1","created_at":"2026-07-05T09:33:15.640409+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.06048","created_at":"2026-07-05T09:33:15.640409+00:00"},{"alias_kind":"pith_short_12","alias_value":"GQ4LX5SKKYOQ","created_at":"2026-07-05T09:33:15.640409+00:00"},{"alias_kind":"pith_short_16","alias_value":"GQ4LX5SKKYOQHF2B","created_at":"2026-07-05T09:33:15.640409+00:00"},{"alias_kind":"pith_short_8","alias_value":"GQ4LX5SK","created_at":"2026-07-05T09:33:15.640409+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22694","citing_title":"SATURN: Symbolic Spatial Reasoning for Multi-Perspective Grounding","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05966","citing_title":"Causal Scaffolding for Physical Reasoning: A Benchmark for Causally-Informed Physical World Understanding in VLMs","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23176","citing_title":"DRIVESPATIAL: A Benchmark for Spatiotemporal Intelligence in VLMs for Autonomous Driving","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23176","citing_title":"DRIVESPATIAL: A Benchmark for Spatiotemporal Intelligence in VLMs for Autonomous Driving","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2601.22228","citing_title":"Lost in Space? Vision-Language Models Struggle with Relative Camera Pose Estimation","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GQ4LX5SKKYOQHF2BFT62VYUCCP","json":"https://pith.science/pith/GQ4LX5SKKYOQHF2BFT62VYUCCP.json","graph_json":"https://pith.science/api/pith-number/GQ4LX5SKKYOQHF2BFT62VYUCCP/graph.json","events_json":"https://pith.science/api/pith-number/GQ4LX5SKKYOQHF2BFT62VYUCCP/events.json","paper":"https://pith.science/paper/GQ4LX5SK"},"agent_actions":{"view_html":"https://pith.science/pith/GQ4LX5SKKYOQHF2BFT62VYUCCP","download_json":"https://pith.science/pith/GQ4LX5SKKYOQHF2BFT62VYUCCP.json","view_paper":"https://pith.science/paper/GQ4LX5SK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.06048&json=true","fetch_graph":"https://pith.science/api/pith-number/GQ4LX5SKKYOQHF2BFT62VYUCCP/graph.json","fetch_events":"https://pith.science/api/pith-number/GQ4LX5SKKYOQHF2BFT62VYUCCP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GQ4LX5SKKYOQHF2BFT62VYUCCP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GQ4LX5SKKYOQHF2BFT62VYUCCP/action/storage_attestation","attest_author":"https://pith.science/pith/GQ4LX5SKKYOQHF2BFT62VYUCCP/action/author_attestation","sign_citation":"https://pith.science/pith/GQ4LX5SKKYOQHF2BFT62VYUCCP/action/citation_signature","submit_replication":"https://pith.science/pith/GQ4LX5SKKYOQHF2BFT62VYUCCP/action/replication_record"}},"created_at":"2026-07-05T09:33:15.640409+00:00","updated_at":"2026-07-05T09:33:15.640409+00:00"}