{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:3MMJFBHKCA4KNDZTLBB4FL5KAK","short_pith_number":"pith:3MMJFBHK","schema_version":"1.0","canonical_sha256":"db189284ea1038a68f335843c2afaa02a5677de0e94bf3473ec556f3c590e137","source":{"kind":"arxiv","id":"2509.02129","version":1},"attestation_state":"computed","paper":{"title":"Scale, Don't Fine-tune: Guiding Multimodal LLMs for Efficient Visual Place Recognition at Test-Time","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.LG","authors_text":"Jiehao Luo, Jintao Cheng, Jin Wu, Weibin Li, Wei Zhang, Xiaoyu Tang, Yao Zou, Zhijian He","submitted_at":"2025-09-02T09:25:13Z","abstract_excerpt":"Visual Place Recognition (VPR) has evolved from handcrafted descriptors to deep learning approaches, yet significant challenges remain. Current approaches, including Vision Foundation Models (VFMs) and Multimodal Large Language Models (MLLMs), enhance semantic understanding but suffer from high computational overhead and limited cross-domain transferability when fine-tuned. To address these limitations, we propose a novel zero-shot framework employing Test-Time Scaling (TTS) that leverages MLLMs' vision-language alignment capabilities through Guidance-based methods for direct similarity scorin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.02129","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-09-02T09:25:13Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"5081c331ee3bdf57dc4290224c910e6d7d23ed91479e7188cef51d8c8393046b","abstract_canon_sha256":"2b3e68e70cfd58ac943d1c29f94df05c009faf29bd5ab6405e9076debb19c111"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:03:15.183464Z","signature_b64":"hk/jDNAi+IFgkZ8LK7+AmdgjBrriMjdlfCpd2jNO7zlalV6Tgc9T35q5p5B3Or7JeoieW+CwAiLXIf5Q9pFMBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"db189284ea1038a68f335843c2afaa02a5677de0e94bf3473ec556f3c590e137","last_reissued_at":"2026-07-05T12:03:15.182954Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:03:15.182954Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scale, Don't Fine-tune: Guiding Multimodal LLMs for Efficient Visual Place Recognition at Test-Time","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.LG","authors_text":"Jiehao Luo, Jintao Cheng, Jin Wu, Weibin Li, Wei Zhang, Xiaoyu Tang, Yao Zou, Zhijian He","submitted_at":"2025-09-02T09:25:13Z","abstract_excerpt":"Visual Place Recognition (VPR) has evolved from handcrafted descriptors to deep learning approaches, yet significant challenges remain. Current approaches, including Vision Foundation Models (VFMs) and Multimodal Large Language Models (MLLMs), enhance semantic understanding but suffer from high computational overhead and limited cross-domain transferability when fine-tuned. To address these limitations, we propose a novel zero-shot framework employing Test-Time Scaling (TTS) that leverages MLLMs' vision-language alignment capabilities through Guidance-based methods for direct similarity scorin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.02129","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.02129/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.02129","created_at":"2026-07-05T12:03:15.183046+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.02129v1","created_at":"2026-07-05T12:03:15.183046+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.02129","created_at":"2026-07-05T12:03:15.183046+00:00"},{"alias_kind":"pith_short_12","alias_value":"3MMJFBHKCA4K","created_at":"2026-07-05T12:03:15.183046+00:00"},{"alias_kind":"pith_short_16","alias_value":"3MMJFBHKCA4KNDZT","created_at":"2026-07-05T12:03:15.183046+00:00"},{"alias_kind":"pith_short_8","alias_value":"3MMJFBHK","created_at":"2026-07-05T12:03:15.183046+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3MMJFBHKCA4KNDZTLBB4FL5KAK","json":"https://pith.science/pith/3MMJFBHKCA4KNDZTLBB4FL5KAK.json","graph_json":"https://pith.science/api/pith-number/3MMJFBHKCA4KNDZTLBB4FL5KAK/graph.json","events_json":"https://pith.science/api/pith-number/3MMJFBHKCA4KNDZTLBB4FL5KAK/events.json","paper":"https://pith.science/paper/3MMJFBHK"},"agent_actions":{"view_html":"https://pith.science/pith/3MMJFBHKCA4KNDZTLBB4FL5KAK","download_json":"https://pith.science/pith/3MMJFBHKCA4KNDZTLBB4FL5KAK.json","view_paper":"https://pith.science/paper/3MMJFBHK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.02129&json=true","fetch_graph":"https://pith.science/api/pith-number/3MMJFBHKCA4KNDZTLBB4FL5KAK/graph.json","fetch_events":"https://pith.science/api/pith-number/3MMJFBHKCA4KNDZTLBB4FL5KAK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3MMJFBHKCA4KNDZTLBB4FL5KAK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3MMJFBHKCA4KNDZTLBB4FL5KAK/action/storage_attestation","attest_author":"https://pith.science/pith/3MMJFBHKCA4KNDZTLBB4FL5KAK/action/author_attestation","sign_citation":"https://pith.science/pith/3MMJFBHKCA4KNDZTLBB4FL5KAK/action/citation_signature","submit_replication":"https://pith.science/pith/3MMJFBHKCA4KNDZTLBB4FL5KAK/action/replication_record"}},"created_at":"2026-07-05T12:03:15.183046+00:00","updated_at":"2026-07-05T12:03:15.183046+00:00"}