{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FXCMPHF65SVBOUPY4LCB5PMMZ5","short_pith_number":"pith:FXCMPHF6","schema_version":"1.0","canonical_sha256":"2dc4c79cbeecaa1751f8e2c41ebd8ccf4d7d4c1b8f2225eb06bfa03fc2862c0c","source":{"kind":"arxiv","id":"2502.11859","version":2},"attestation_state":"computed","paper":{"title":"Defining and Evaluating Visual Language Models' Basic Spatial Abilities: A Perspective from Psychometrics","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Chen Gao, Dalin Lyu, Jie Feng, Weihang Wang, Wenrui Xu, Yong Li","submitted_at":"2025-02-17T14:50:53Z","abstract_excerpt":"The Theory of Multiple Intelligences underscores the hierarchical nature of cognitive capabilities. To advance Spatial Artificial Intelligence, we pioneer a psychometric framework defining five Basic Spatial Abilities (BSAs) in Visual Language Models (VLMs): Spatial Perception, Spatial Relation, Spatial Orientation, Mental Rotation, and Spatial Visualization. Benchmarking 13 mainstream VLMs through nine validated psychometric experiments reveals significant gaps versus humans (average score 24.95 vs. 68.38), with three key findings: 1) VLMs mirror human hierarchies (strongest in 2D orientation"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.11859","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-02-17T14:50:53Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"d0e3e31e003d667cfeb2fb6eff79a91a2069b99461cbde9fdb5935408b9fd068","abstract_canon_sha256":"3cef1eb9d428ca767ec5c5b2c250d68fec1b9e250bc90bd9430acd8637da29dc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:48:23.030725Z","signature_b64":"GqSd+wRaUNMLY2LRKvE80K7WX17LE211N/tG/wAgL593PsXfVNQm83gBrLCEFPjV4vE6yZnyAvfawoYsaKZTCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2dc4c79cbeecaa1751f8e2c41ebd8ccf4d7d4c1b8f2225eb06bfa03fc2862c0c","last_reissued_at":"2026-07-05T11:48:23.030207Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:48:23.030207Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Defining and Evaluating Visual Language Models' Basic Spatial Abilities: A Perspective from Psychometrics","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Chen Gao, Dalin Lyu, Jie Feng, Weihang Wang, Wenrui Xu, Yong Li","submitted_at":"2025-02-17T14:50:53Z","abstract_excerpt":"The Theory of Multiple Intelligences underscores the hierarchical nature of cognitive capabilities. To advance Spatial Artificial Intelligence, we pioneer a psychometric framework defining five Basic Spatial Abilities (BSAs) in Visual Language Models (VLMs): Spatial Perception, Spatial Relation, Spatial Orientation, Mental Rotation, and Spatial Visualization. Benchmarking 13 mainstream VLMs through nine validated psychometric experiments reveals significant gaps versus humans (average score 24.95 vs. 68.38), with three key findings: 1) VLMs mirror human hierarchies (strongest in 2D orientation"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.11859","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.11859/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.11859","created_at":"2026-07-05T11:48:23.030265+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.11859v2","created_at":"2026-07-05T11:48:23.030265+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.11859","created_at":"2026-07-05T11:48:23.030265+00:00"},{"alias_kind":"pith_short_12","alias_value":"FXCMPHF65SVB","created_at":"2026-07-05T11:48:23.030265+00:00"},{"alias_kind":"pith_short_16","alias_value":"FXCMPHF65SVBOUPY","created_at":"2026-07-05T11:48:23.030265+00:00"},{"alias_kind":"pith_short_8","alias_value":"FXCMPHF6","created_at":"2026-07-05T11:48:23.030265+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.20068","citing_title":"11Plus-Bench: Demystifying Multimodal LLM Spatial Reasoning with Cognitive-Inspired Analysis","ref_index":72,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FXCMPHF65SVBOUPY4LCB5PMMZ5","json":"https://pith.science/pith/FXCMPHF65SVBOUPY4LCB5PMMZ5.json","graph_json":"https://pith.science/api/pith-number/FXCMPHF65SVBOUPY4LCB5PMMZ5/graph.json","events_json":"https://pith.science/api/pith-number/FXCMPHF65SVBOUPY4LCB5PMMZ5/events.json","paper":"https://pith.science/paper/FXCMPHF6"},"agent_actions":{"view_html":"https://pith.science/pith/FXCMPHF65SVBOUPY4LCB5PMMZ5","download_json":"https://pith.science/pith/FXCMPHF65SVBOUPY4LCB5PMMZ5.json","view_paper":"https://pith.science/paper/FXCMPHF6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.11859&json=true","fetch_graph":"https://pith.science/api/pith-number/FXCMPHF65SVBOUPY4LCB5PMMZ5/graph.json","fetch_events":"https://pith.science/api/pith-number/FXCMPHF65SVBOUPY4LCB5PMMZ5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FXCMPHF65SVBOUPY4LCB5PMMZ5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FXCMPHF65SVBOUPY4LCB5PMMZ5/action/storage_attestation","attest_author":"https://pith.science/pith/FXCMPHF65SVBOUPY4LCB5PMMZ5/action/author_attestation","sign_citation":"https://pith.science/pith/FXCMPHF65SVBOUPY4LCB5PMMZ5/action/citation_signature","submit_replication":"https://pith.science/pith/FXCMPHF65SVBOUPY4LCB5PMMZ5/action/replication_record"}},"created_at":"2026-07-05T11:48:23.030265+00:00","updated_at":"2026-07-05T11:48:23.030265+00:00"}