{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:76FHTYSGXX57BMSN6BMFMDVSYJ","short_pith_number":"pith:76FHTYSG","schema_version":"1.0","canonical_sha256":"ff8a79e246bdfbf0b24df058560eb2c2457b2702279b03668174f06836e70a33","source":{"kind":"arxiv","id":"2410.17385","version":2},"attestation_state":"computed","paper":{"title":"Do Vision-Language Models Represent Space and How? Evaluating Spatial Frame of Reference Under Ambiguities","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Fengyuan Hu, Freda Shi, Jayjun Lee, Joyce Chai, Parisa Kordjamshidi, Zheyuan Zhang, Ziqiao Ma","submitted_at":"2024-10-22T19:39:15Z","abstract_excerpt":"Spatial expressions in situated communication can be ambiguous, as their meanings vary depending on the frames of reference (FoR) adopted by speakers and listeners. While spatial language understanding and reasoning by vision-language models (VLMs) have gained increasing attention, potential ambiguities in these models are still under-explored. To address this issue, we present the COnsistent Multilingual Frame Of Reference Test (COMFORT), an evaluation protocol to systematically assess the spatial reasoning capabilities of VLMs. We evaluate nine state-of-the-art VLMs using COMFORT. Despite sh"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.17385","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-22T19:39:15Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"9113e4d2d5027addc1f8a017701be7f7b757be6694ecdbd643740e768bc95055","abstract_canon_sha256":"56acbaf0eb2f4fc76f146a37b1f78e858d864e4e8839c5a40c920dd1f033a115"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:50:31.278429Z","signature_b64":"0sLxhRSGUY3gERttwOl+OkuD/UWiVNoKzZUCFKeW50M3StrqIX1BGHvFqh7WH0KNnBBXuxZ7KtNKO8FctbDPAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ff8a79e246bdfbf0b24df058560eb2c2457b2702279b03668174f06836e70a33","last_reissued_at":"2026-07-05T10:50:31.277791Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:50:31.277791Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Do Vision-Language Models Represent Space and How? Evaluating Spatial Frame of Reference Under Ambiguities","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Fengyuan Hu, Freda Shi, Jayjun Lee, Joyce Chai, Parisa Kordjamshidi, Zheyuan Zhang, Ziqiao Ma","submitted_at":"2024-10-22T19:39:15Z","abstract_excerpt":"Spatial expressions in situated communication can be ambiguous, as their meanings vary depending on the frames of reference (FoR) adopted by speakers and listeners. While spatial language understanding and reasoning by vision-language models (VLMs) have gained increasing attention, potential ambiguities in these models are still under-explored. To address this issue, we present the COnsistent Multilingual Frame Of Reference Test (COMFORT), an evaluation protocol to systematically assess the spatial reasoning capabilities of VLMs. We evaluate nine state-of-the-art VLMs using COMFORT. Despite sh"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.17385","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.17385/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.17385","created_at":"2026-07-05T10:50:31.277854+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.17385v2","created_at":"2026-07-05T10:50:31.277854+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.17385","created_at":"2026-07-05T10:50:31.277854+00:00"},{"alias_kind":"pith_short_12","alias_value":"76FHTYSGXX57","created_at":"2026-07-05T10:50:31.277854+00:00"},{"alias_kind":"pith_short_16","alias_value":"76FHTYSGXX57BMSN","created_at":"2026-07-05T10:50:31.277854+00:00"},{"alias_kind":"pith_short_8","alias_value":"76FHTYSG","created_at":"2026-07-05T10:50:31.277854+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08029","citing_title":"IntentNav: Learning Spatial-Visual Object Navigation from Human Demonstrations","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00881","citing_title":"OmniView-Space: Reinforcing Spatial Reasoning via Multi-Perspective Spatial Mapping","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00273","citing_title":"When Do Diffusion Models learn to Generate Multiple Objects?","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25524","citing_title":"ProSR: Process-Shaped Spatial Reasoning for Reliable Chain-of-Thought in VLMs","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2503.07557","citing_title":"AutoSpatial: Visual-Language Reasoning for Social Robot Navigation through Efficient Spatial Reasoning Learning","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/76FHTYSGXX57BMSN6BMFMDVSYJ","json":"https://pith.science/pith/76FHTYSGXX57BMSN6BMFMDVSYJ.json","graph_json":"https://pith.science/api/pith-number/76FHTYSGXX57BMSN6BMFMDVSYJ/graph.json","events_json":"https://pith.science/api/pith-number/76FHTYSGXX57BMSN6BMFMDVSYJ/events.json","paper":"https://pith.science/paper/76FHTYSG"},"agent_actions":{"view_html":"https://pith.science/pith/76FHTYSGXX57BMSN6BMFMDVSYJ","download_json":"https://pith.science/pith/76FHTYSGXX57BMSN6BMFMDVSYJ.json","view_paper":"https://pith.science/paper/76FHTYSG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.17385&json=true","fetch_graph":"https://pith.science/api/pith-number/76FHTYSGXX57BMSN6BMFMDVSYJ/graph.json","fetch_events":"https://pith.science/api/pith-number/76FHTYSGXX57BMSN6BMFMDVSYJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/76FHTYSGXX57BMSN6BMFMDVSYJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/76FHTYSGXX57BMSN6BMFMDVSYJ/action/storage_attestation","attest_author":"https://pith.science/pith/76FHTYSGXX57BMSN6BMFMDVSYJ/action/author_attestation","sign_citation":"https://pith.science/pith/76FHTYSGXX57BMSN6BMFMDVSYJ/action/citation_signature","submit_replication":"https://pith.science/pith/76FHTYSGXX57BMSN6BMFMDVSYJ/action/replication_record"}},"created_at":"2026-07-05T10:50:31.277854+00:00","updated_at":"2026-07-05T10:50:31.277854+00:00"}