{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:HWULDXXEUZS5BE7SXZAVPSRYJL","short_pith_number":"pith:HWULDXXE","schema_version":"1.0","canonical_sha256":"3da8b1dee4a665d093f2be4157ca384ae6449dd18663133a3dbff66d1025f860","source":{"kind":"arxiv","id":"2502.14638","version":1},"attestation_state":"computed","paper":{"title":"NAVIG: Natural Language-guided Analysis with Vision Language Models for Image Geo-localization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Jordan Boyd-Graber, Runze Li, Tasnim Kabir, Zheyuan Zhang","submitted_at":"2025-02-20T15:21:35Z","abstract_excerpt":"Image geo-localization is the task of predicting the specific location of an image and requires complex reasoning across visual, geographical, and cultural contexts. While prior Vision Language Models (VLMs) have the best accuracy at this task, there is a dearth of high-quality datasets and models for analytical reasoning. We first create NaviClues, a high-quality dataset derived from GeoGuessr, a popular geography game, to supply examples of expert reasoning from language. Using this dataset, we present Navig, a comprehensive image geo-localization framework integrating global and fine-graine"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.14638","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-02-20T15:21:35Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"e3c328ba8206610c42a20ab4db6c2915033ec3812c26e1ed21a815103f4a26c7","abstract_canon_sha256":"21b7a2003b474d8a818afa9b90573e10889c13efae46dfc1108335b4adb8d50e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:17:30.482499Z","signature_b64":"0BXNLKUEcUUX5NqyXr5K+kzVXlvdu7P0ndLeo+UKN1BnqGwxA5ubg6ReTh6aXYLzNO6oa+xcAeiJS77J3D6kAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3da8b1dee4a665d093f2be4157ca384ae6449dd18663133a3dbff66d1025f860","last_reissued_at":"2026-07-05T10:17:30.482034Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:17:30.482034Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"NAVIG: Natural Language-guided Analysis with Vision Language Models for Image Geo-localization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Jordan Boyd-Graber, Runze Li, Tasnim Kabir, Zheyuan Zhang","submitted_at":"2025-02-20T15:21:35Z","abstract_excerpt":"Image geo-localization is the task of predicting the specific location of an image and requires complex reasoning across visual, geographical, and cultural contexts. While prior Vision Language Models (VLMs) have the best accuracy at this task, there is a dearth of high-quality datasets and models for analytical reasoning. We first create NaviClues, a high-quality dataset derived from GeoGuessr, a popular geography game, to supply examples of expert reasoning from language. Using this dataset, we present Navig, a comprehensive image geo-localization framework integrating global and fine-graine"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.14638","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.14638/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.14638","created_at":"2026-07-05T10:17:30.482089+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.14638v1","created_at":"2026-07-05T10:17:30.482089+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.14638","created_at":"2026-07-05T10:17:30.482089+00:00"},{"alias_kind":"pith_short_12","alias_value":"HWULDXXEUZS5","created_at":"2026-07-05T10:17:30.482089+00:00"},{"alias_kind":"pith_short_16","alias_value":"HWULDXXEUZS5BE7S","created_at":"2026-07-05T10:17:30.482089+00:00"},{"alias_kind":"pith_short_8","alias_value":"HWULDXXE","created_at":"2026-07-05T10:17:30.482089+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.02129","citing_title":"Scale, Don't Fine-tune: Guiding Multimodal LLMs for Efficient Visual Place Recognition at Test-Time","ref_index":26,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HWULDXXEUZS5BE7SXZAVPSRYJL","json":"https://pith.science/pith/HWULDXXEUZS5BE7SXZAVPSRYJL.json","graph_json":"https://pith.science/api/pith-number/HWULDXXEUZS5BE7SXZAVPSRYJL/graph.json","events_json":"https://pith.science/api/pith-number/HWULDXXEUZS5BE7SXZAVPSRYJL/events.json","paper":"https://pith.science/paper/HWULDXXE"},"agent_actions":{"view_html":"https://pith.science/pith/HWULDXXEUZS5BE7SXZAVPSRYJL","download_json":"https://pith.science/pith/HWULDXXEUZS5BE7SXZAVPSRYJL.json","view_paper":"https://pith.science/paper/HWULDXXE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.14638&json=true","fetch_graph":"https://pith.science/api/pith-number/HWULDXXEUZS5BE7SXZAVPSRYJL/graph.json","fetch_events":"https://pith.science/api/pith-number/HWULDXXEUZS5BE7SXZAVPSRYJL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HWULDXXEUZS5BE7SXZAVPSRYJL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HWULDXXEUZS5BE7SXZAVPSRYJL/action/storage_attestation","attest_author":"https://pith.science/pith/HWULDXXEUZS5BE7SXZAVPSRYJL/action/author_attestation","sign_citation":"https://pith.science/pith/HWULDXXEUZS5BE7SXZAVPSRYJL/action/citation_signature","submit_replication":"https://pith.science/pith/HWULDXXEUZS5BE7SXZAVPSRYJL/action/replication_record"}},"created_at":"2026-07-05T10:17:30.482089+00:00","updated_at":"2026-07-05T10:17:30.482089+00:00"}