{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:UVEWP2TXSEF5IBIG6YL3N3DFSV","short_pith_number":"pith:UVEWP2TX","schema_version":"1.0","canonical_sha256":"a54967ea77910bd40506f617b6ec659543c0de10e55242aea310ee8b40911afb","source":{"kind":"arxiv","id":"2502.14412","version":1},"attestation_state":"computed","paper":{"title":"Evaluating Precise Geolocation Inference Capabilities of Vision Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CR","cs.LG"],"primary_cat":"cs.CV","authors_text":"Hieu Minh Nguyen, Jacob Haimes, Neel Jay, Trung Dung Hoang","submitted_at":"2025-02-20T09:59:28Z","abstract_excerpt":"The prevalence of Vision-Language Models (VLMs) raises important questions about privacy in an era where visual information is increasingly available. While foundation VLMs demonstrate broad knowledge and learned capabilities, we specifically investigate their ability to infer geographic location from previously unseen image data. This paper introduces a benchmark dataset collected from Google Street View that represents its global distribution of coverage. Foundation models are evaluated on single-image geolocation inference, with many achieving median distance errors of <300 km. We further e"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.14412","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-02-20T09:59:28Z","cross_cats_sorted":["cs.CR","cs.LG"],"title_canon_sha256":"f34c1a6e6be569b51a3bcf0cae9d4824f9f9ca1efd73ff61ddc6cb0f8be55eef","abstract_canon_sha256":"b2046502cfcdde6af2eef9ba22d8816004272a8bf21d8344b012cfc9dd29fc1c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:17:25.953200Z","signature_b64":"NdmH9SrIssHpdGzfvSs80qsTOOJ1TSghAfNQ5ZnbNMO5NB0bo3lM6dBByP5NjuKqYpYOJSlJt9IKHPSNVjb0Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a54967ea77910bd40506f617b6ec659543c0de10e55242aea310ee8b40911afb","last_reissued_at":"2026-07-05T10:17:25.952711Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:17:25.952711Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating Precise Geolocation Inference Capabilities of Vision Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CR","cs.LG"],"primary_cat":"cs.CV","authors_text":"Hieu Minh Nguyen, Jacob Haimes, Neel Jay, Trung Dung Hoang","submitted_at":"2025-02-20T09:59:28Z","abstract_excerpt":"The prevalence of Vision-Language Models (VLMs) raises important questions about privacy in an era where visual information is increasingly available. While foundation VLMs demonstrate broad knowledge and learned capabilities, we specifically investigate their ability to infer geographic location from previously unseen image data. This paper introduces a benchmark dataset collected from Google Street View that represents its global distribution of coverage. Foundation models are evaluated on single-image geolocation inference, with many achieving median distance errors of <300 km. We further e"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.14412","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.14412/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.14412","created_at":"2026-07-05T10:17:25.952771+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.14412v1","created_at":"2026-07-05T10:17:25.952771+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.14412","created_at":"2026-07-05T10:17:25.952771+00:00"},{"alias_kind":"pith_short_12","alias_value":"UVEWP2TXSEF5","created_at":"2026-07-05T10:17:25.952771+00:00"},{"alias_kind":"pith_short_16","alias_value":"UVEWP2TXSEF5IBIG","created_at":"2026-07-05T10:17:25.952771+00:00"},{"alias_kind":"pith_short_8","alias_value":"UVEWP2TX","created_at":"2026-07-05T10:17:25.952771+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07642","citing_title":"Do VLMs See What Sensors Feel? A Scalable Expert-Guided Design for Wheelchair Accessibility Assessment from Street View","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2508.01608","citing_title":"From Pixels to Places: A Systematic Benchmark for Evaluating Image Geolocalization Ability in Large Language Models","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UVEWP2TXSEF5IBIG6YL3N3DFSV","json":"https://pith.science/pith/UVEWP2TXSEF5IBIG6YL3N3DFSV.json","graph_json":"https://pith.science/api/pith-number/UVEWP2TXSEF5IBIG6YL3N3DFSV/graph.json","events_json":"https://pith.science/api/pith-number/UVEWP2TXSEF5IBIG6YL3N3DFSV/events.json","paper":"https://pith.science/paper/UVEWP2TX"},"agent_actions":{"view_html":"https://pith.science/pith/UVEWP2TXSEF5IBIG6YL3N3DFSV","download_json":"https://pith.science/pith/UVEWP2TXSEF5IBIG6YL3N3DFSV.json","view_paper":"https://pith.science/paper/UVEWP2TX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.14412&json=true","fetch_graph":"https://pith.science/api/pith-number/UVEWP2TXSEF5IBIG6YL3N3DFSV/graph.json","fetch_events":"https://pith.science/api/pith-number/UVEWP2TXSEF5IBIG6YL3N3DFSV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UVEWP2TXSEF5IBIG6YL3N3DFSV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UVEWP2TXSEF5IBIG6YL3N3DFSV/action/storage_attestation","attest_author":"https://pith.science/pith/UVEWP2TXSEF5IBIG6YL3N3DFSV/action/author_attestation","sign_citation":"https://pith.science/pith/UVEWP2TXSEF5IBIG6YL3N3DFSV/action/citation_signature","submit_replication":"https://pith.science/pith/UVEWP2TXSEF5IBIG6YL3N3DFSV/action/replication_record"}},"created_at":"2026-07-05T10:17:25.952771+00:00","updated_at":"2026-07-05T10:17:25.952771+00:00"}