{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KOA7AP3GB2CGFFRHDMASMS2JVQ","short_pith_number":"pith:KOA7AP3G","schema_version":"1.0","canonical_sha256":"5381f03f660e846296271b01264b49ac2a72467de9b3a244097ee61006728a8f","source":{"kind":"arxiv","id":"2407.06730","version":1},"attestation_state":"computed","paper":{"title":"LVLM-empowered Multi-modal Representation Learning for Visual Place Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Changyin Sun, Lei Cheng, Lingquan Meng, Teng Wang","submitted_at":"2024-07-09T10:15:31Z","abstract_excerpt":"Visual place recognition (VPR) remains challenging due to significant viewpoint changes and appearance variations. Mainstream works tackle these challenges by developing various feature aggregation methods to transform deep features into robust and compact global representations. Unfortunately, satisfactory results cannot be achieved under challenging conditions. We start from a new perspective and attempt to build a discriminative global representations by fusing image data and text descriptions of the the visual scene. The motivation is twofold: (1) Current Large Vision-Language Models (LVLM"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.06730","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-07-09T10:15:31Z","cross_cats_sorted":[],"title_canon_sha256":"a9a6822616fca6bde84c177d8d72d4dafbf8ff0363036a5737ea7a538eae9b0e","abstract_canon_sha256":"36a057eaac4359fb6ca294bfadfd6011145f4300ca90c80214728cfd407e3a00"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:41:52.622213Z","signature_b64":"lvVcZFmTd6IJugffLeLb0biyhaMRVLQuTodKl6ftB4rsXAUZ66w55J58Wh0+1IzR2ou7eWQEG1uP2ZG//nX5Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5381f03f660e846296271b01264b49ac2a72467de9b3a244097ee61006728a8f","last_reissued_at":"2026-07-05T08:41:52.621812Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:41:52.621812Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LVLM-empowered Multi-modal Representation Learning for Visual Place Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Changyin Sun, Lei Cheng, Lingquan Meng, Teng Wang","submitted_at":"2024-07-09T10:15:31Z","abstract_excerpt":"Visual place recognition (VPR) remains challenging due to significant viewpoint changes and appearance variations. Mainstream works tackle these challenges by developing various feature aggregation methods to transform deep features into robust and compact global representations. Unfortunately, satisfactory results cannot be achieved under challenging conditions. We start from a new perspective and attempt to build a discriminative global representations by fusing image data and text descriptions of the the visual scene. The motivation is twofold: (1) Current Large Vision-Language Models (LVLM"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.06730","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.06730/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.06730","created_at":"2026-07-05T08:41:52.621873+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.06730v1","created_at":"2026-07-05T08:41:52.621873+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.06730","created_at":"2026-07-05T08:41:52.621873+00:00"},{"alias_kind":"pith_short_12","alias_value":"KOA7AP3GB2CG","created_at":"2026-07-05T08:41:52.621873+00:00"},{"alias_kind":"pith_short_16","alias_value":"KOA7AP3GB2CGFFRH","created_at":"2026-07-05T08:41:52.621873+00:00"},{"alias_kind":"pith_short_8","alias_value":"KOA7AP3G","created_at":"2026-07-05T08:41:52.621873+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KOA7AP3GB2CGFFRHDMASMS2JVQ","json":"https://pith.science/pith/KOA7AP3GB2CGFFRHDMASMS2JVQ.json","graph_json":"https://pith.science/api/pith-number/KOA7AP3GB2CGFFRHDMASMS2JVQ/graph.json","events_json":"https://pith.science/api/pith-number/KOA7AP3GB2CGFFRHDMASMS2JVQ/events.json","paper":"https://pith.science/paper/KOA7AP3G"},"agent_actions":{"view_html":"https://pith.science/pith/KOA7AP3GB2CGFFRHDMASMS2JVQ","download_json":"https://pith.science/pith/KOA7AP3GB2CGFFRHDMASMS2JVQ.json","view_paper":"https://pith.science/paper/KOA7AP3G","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.06730&json=true","fetch_graph":"https://pith.science/api/pith-number/KOA7AP3GB2CGFFRHDMASMS2JVQ/graph.json","fetch_events":"https://pith.science/api/pith-number/KOA7AP3GB2CGFFRHDMASMS2JVQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KOA7AP3GB2CGFFRHDMASMS2JVQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KOA7AP3GB2CGFFRHDMASMS2JVQ/action/storage_attestation","attest_author":"https://pith.science/pith/KOA7AP3GB2CGFFRHDMASMS2JVQ/action/author_attestation","sign_citation":"https://pith.science/pith/KOA7AP3GB2CGFFRHDMASMS2JVQ/action/citation_signature","submit_replication":"https://pith.science/pith/KOA7AP3GB2CGFFRHDMASMS2JVQ/action/replication_record"}},"created_at":"2026-07-05T08:41:52.621873+00:00","updated_at":"2026-07-05T08:41:52.621873+00:00"}