{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:USBM4ZIIUA3YFJHFSA6OCFLGEK","short_pith_number":"pith:USBM4ZII","schema_version":"1.0","canonical_sha256":"a482ce6508a03782a4e5903ce1156622b48776e18b31c3646a109fa68e1cc2c3","source":{"kind":"arxiv","id":"2402.11095","version":1},"attestation_state":"computed","paper":{"title":"GIM: Learning Generalizable Image Matcher From Internet Videos","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Cheng Wang, Kaixuan Wang, Matthias M\\\"uller, Wei Yin, Xiaozhi Chen, Xuelun Shen, Zhipeng Cai, Zijun Li","submitted_at":"2024-02-16T21:48:17Z","abstract_excerpt":"Image matching is a fundamental computer vision problem. While learning-based methods achieve state-of-the-art performance on existing benchmarks, they generalize poorly to in-the-wild images. Such methods typically need to train separate models for different scene types and are impractical when the scene type is unknown in advance. One of the underlying problems is the limited scalability of existing data construction pipelines, which limits the diversity of standard image matching datasets. To address this problem, we propose GIM, a self-training framework for learning a single generalizable"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.11095","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-02-16T21:48:17Z","cross_cats_sorted":[],"title_canon_sha256":"7a3790056f57865f8a21302d940aa84d3919c0e7c2edaf0ecb3eecfb93801524","abstract_canon_sha256":"d58a32c39bb01dacb5e642c3c9fd69866ce9b1707e7a8bdba420af45e378f1bc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:46:34.107899Z","signature_b64":"m+jz1RGrxRpGJbTIJ/LJ85SkxMVLg4uQhOcnxPnahURL33CQCVBP/qivjXxM/6JFHrgnBB5ef4cNYxjRYFthDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a482ce6508a03782a4e5903ce1156622b48776e18b31c3646a109fa68e1cc2c3","last_reissued_at":"2026-07-05T07:46:34.107396Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:46:34.107396Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GIM: Learning Generalizable Image Matcher From Internet Videos","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Cheng Wang, Kaixuan Wang, Matthias M\\\"uller, Wei Yin, Xiaozhi Chen, Xuelun Shen, Zhipeng Cai, Zijun Li","submitted_at":"2024-02-16T21:48:17Z","abstract_excerpt":"Image matching is a fundamental computer vision problem. While learning-based methods achieve state-of-the-art performance on existing benchmarks, they generalize poorly to in-the-wild images. Such methods typically need to train separate models for different scene types and are impractical when the scene type is unknown in advance. One of the underlying problems is the limited scalability of existing data construction pipelines, which limits the diversity of standard image matching datasets. To address this problem, we propose GIM, a self-training framework for learning a single generalizable"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.11095","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.11095/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.11095","created_at":"2026-07-05T07:46:34.107458+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.11095v1","created_at":"2026-07-05T07:46:34.107458+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.11095","created_at":"2026-07-05T07:46:34.107458+00:00"},{"alias_kind":"pith_short_12","alias_value":"USBM4ZIIUA3Y","created_at":"2026-07-05T07:46:34.107458+00:00"},{"alias_kind":"pith_short_16","alias_value":"USBM4ZIIUA3YFJHF","created_at":"2026-07-05T07:46:34.107458+00:00"},{"alias_kind":"pith_short_8","alias_value":"USBM4ZII","created_at":"2026-07-05T07:46:34.107458+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05376","citing_title":"MV-Forcing: Long Multi-View Video Generation via 4D-Grounded Spatio-Temporal Self-Forcing","ref_index":29,"is_internal_anchor":true},{"citing_arxiv_id":"2607.01748","citing_title":"RTE-FM-Dehazer: Radiative Transfer Equation Inspired Flow Matching for Real-World Image Dehazing","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07326","citing_title":"AnchorWorld: Embodied Egocentric World Simulation with View-based Evolution Customization","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30561","citing_title":"VLM3: Vision Language Models Are Native 3D Learners","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03377","citing_title":"ViBA: Implicit Bundle Adjustment with Geometric and Temporal Consistency for Robust Visual Matching","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21776","citing_title":"Reshoot-Anything: A Self-Supervised Model for In-the-Wild Video Reshooting","ref_index":30,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/USBM4ZIIUA3YFJHFSA6OCFLGEK","json":"https://pith.science/pith/USBM4ZIIUA3YFJHFSA6OCFLGEK.json","graph_json":"https://pith.science/api/pith-number/USBM4ZIIUA3YFJHFSA6OCFLGEK/graph.json","events_json":"https://pith.science/api/pith-number/USBM4ZIIUA3YFJHFSA6OCFLGEK/events.json","paper":"https://pith.science/paper/USBM4ZII"},"agent_actions":{"view_html":"https://pith.science/pith/USBM4ZIIUA3YFJHFSA6OCFLGEK","download_json":"https://pith.science/pith/USBM4ZIIUA3YFJHFSA6OCFLGEK.json","view_paper":"https://pith.science/paper/USBM4ZII","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.11095&json=true","fetch_graph":"https://pith.science/api/pith-number/USBM4ZIIUA3YFJHFSA6OCFLGEK/graph.json","fetch_events":"https://pith.science/api/pith-number/USBM4ZIIUA3YFJHFSA6OCFLGEK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/USBM4ZIIUA3YFJHFSA6OCFLGEK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/USBM4ZIIUA3YFJHFSA6OCFLGEK/action/storage_attestation","attest_author":"https://pith.science/pith/USBM4ZIIUA3YFJHFSA6OCFLGEK/action/author_attestation","sign_citation":"https://pith.science/pith/USBM4ZIIUA3YFJHFSA6OCFLGEK/action/citation_signature","submit_replication":"https://pith.science/pith/USBM4ZIIUA3YFJHFSA6OCFLGEK/action/replication_record"}},"created_at":"2026-07-05T07:46:34.107458+00:00","updated_at":"2026-07-05T07:46:34.107458+00:00"}