{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:FRLI64OA6CPJKNG236OQTFLJ2P","short_pith_number":"pith:FRLI64OA","schema_version":"1.0","canonical_sha256":"2c568f71c0f09e9534dadf9d099569d3e68c579ba8035036b53d1e12713cb287","source":{"kind":"arxiv","id":"2607.07251","version":1},"attestation_state":"computed","paper":{"title":"Evaluation of Multilingual Ability to Use Spatial Deictic Expressions in Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hitomi Yanaka, Kaito Watanabe, Taisei Yamamoto, Tomoki Doi","submitted_at":"2026-07-08T10:34:54Z","abstract_excerpt":"One of the expected abilities of vision-language models (VLMs) is spatial reasoning ability based on a given text and image. To evaluate the spatial reasoning abilities of VLMs, we focus on the use of spatial deictic expressions, which are defined as spatial expressions whose referent is determined by their situational context, such as ``this'' and ``that''. To handle spatial deictic expressions, VLMs must jointly reason over language and visual space, grounding context-dependent references in the image's spatial structure. In addition, selecting appropriate spatial deictic expressions across "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.07251","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-07-08T10:34:54Z","cross_cats_sorted":[],"title_canon_sha256":"b3580b4a43890cbd55450e3d52cbea41302bb1bc8ff362809814e94b543a4587","abstract_canon_sha256":"417fae9a2cc6ea9542e112ee9bc803a2029a4c27394a405f3abbf212d54ae47e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-09T01:20:18.881245Z","signature_b64":"k/hQ1No5ELTaNzEePRbZ6lnq5tFI6Zli02974d1kQ/RAryTCgNe6ym8dBjibbDnDFPhVELsE5bFUs5zFYcCFBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2c568f71c0f09e9534dadf9d099569d3e68c579ba8035036b53d1e12713cb287","last_reissued_at":"2026-07-09T01:20:18.880842Z","signature_status":"signed_v1","first_computed_at":"2026-07-09T01:20:18.880842Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluation of Multilingual Ability to Use Spatial Deictic Expressions in Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hitomi Yanaka, Kaito Watanabe, Taisei Yamamoto, Tomoki Doi","submitted_at":"2026-07-08T10:34:54Z","abstract_excerpt":"One of the expected abilities of vision-language models (VLMs) is spatial reasoning ability based on a given text and image. To evaluate the spatial reasoning abilities of VLMs, we focus on the use of spatial deictic expressions, which are defined as spatial expressions whose referent is determined by their situational context, such as ``this'' and ``that''. To handle spatial deictic expressions, VLMs must jointly reason over language and visual space, grounding context-dependent references in the image's spatial structure. In addition, selecting appropriate spatial deictic expressions across "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.07251","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.07251/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.07251","created_at":"2026-07-09T01:20:18.880898+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.07251v1","created_at":"2026-07-09T01:20:18.880898+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.07251","created_at":"2026-07-09T01:20:18.880898+00:00"},{"alias_kind":"pith_short_12","alias_value":"FRLI64OA6CPJ","created_at":"2026-07-09T01:20:18.880898+00:00"},{"alias_kind":"pith_short_16","alias_value":"FRLI64OA6CPJKNG2","created_at":"2026-07-09T01:20:18.880898+00:00"},{"alias_kind":"pith_short_8","alias_value":"FRLI64OA","created_at":"2026-07-09T01:20:18.880898+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FRLI64OA6CPJKNG236OQTFLJ2P","json":"https://pith.science/pith/FRLI64OA6CPJKNG236OQTFLJ2P.json","graph_json":"https://pith.science/api/pith-number/FRLI64OA6CPJKNG236OQTFLJ2P/graph.json","events_json":"https://pith.science/api/pith-number/FRLI64OA6CPJKNG236OQTFLJ2P/events.json","paper":"https://pith.science/paper/FRLI64OA"},"agent_actions":{"view_html":"https://pith.science/pith/FRLI64OA6CPJKNG236OQTFLJ2P","download_json":"https://pith.science/pith/FRLI64OA6CPJKNG236OQTFLJ2P.json","view_paper":"https://pith.science/paper/FRLI64OA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.07251&json=true","fetch_graph":"https://pith.science/api/pith-number/FRLI64OA6CPJKNG236OQTFLJ2P/graph.json","fetch_events":"https://pith.science/api/pith-number/FRLI64OA6CPJKNG236OQTFLJ2P/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FRLI64OA6CPJKNG236OQTFLJ2P/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FRLI64OA6CPJKNG236OQTFLJ2P/action/storage_attestation","attest_author":"https://pith.science/pith/FRLI64OA6CPJKNG236OQTFLJ2P/action/author_attestation","sign_citation":"https://pith.science/pith/FRLI64OA6CPJKNG236OQTFLJ2P/action/citation_signature","submit_replication":"https://pith.science/pith/FRLI64OA6CPJKNG236OQTFLJ2P/action/replication_record"}},"created_at":"2026-07-09T01:20:18.880898+00:00","updated_at":"2026-07-09T01:20:18.880898+00:00"}