{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CUAYLEV22OKVYAHIFW2QVJQ65B","short_pith_number":"pith:CUAYLEV2","schema_version":"1.0","canonical_sha256":"15018592bad3955c00e82db50aa61ee85d381d61bcc8e87491c20d31fd9cb8b6","source":{"kind":"arxiv","id":"2501.04670","version":3},"attestation_state":"computed","paper":{"title":"Are They the Same? Exploring Visual Correspondence Shortcomings of Multimodal LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiangning Zhang, Lu Qi, Qianyu Zhou, Shihao Chen, Shilin Xu, Shunping Ji, Tao Zhang, Xiangtai Li, Yikang Zhou, Yunhai Tong","submitted_at":"2025-01-08T18:30:53Z","abstract_excerpt":"Recent advancements in multimodal large language models (MLLM) have shown a strong ability in visual perception, reasoning abilities, and vision-language understanding. However, the visual matching ability of MLLMs is rarely studied, despite finding the visual correspondence of objects is essential in computer vision. Our research reveals that the matching capabilities in recent MLLMs still exhibit systematic shortcomings, even with current strong MLLMs models, GPT-4o. In particular, we construct a Multimodal Visual Matching (MMVM) benchmark to fairly benchmark over 30 different MLLMs. The MMV"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.04670","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-01-08T18:30:53Z","cross_cats_sorted":[],"title_canon_sha256":"a8895f6a19ee5f384818ec120a0c0e6d40e3fce26296cc24d65c54be1cd5ed9b","abstract_canon_sha256":"dc33808379d855750032e625e1b88677c5c0e6ac79c6c61f76513da246014425"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:34:03.493234Z","signature_b64":"vpfSs0kv30jP9ktr9eYZEFsOWIY1OkQMm1SgBGM49wwu/DpT4GjakjOSFv7Pj+BfHfH+NPmpvklUOIUPpt7tBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"15018592bad3955c00e82db50aa61ee85d381d61bcc8e87491c20d31fd9cb8b6","last_reissued_at":"2026-07-05T11:34:03.492709Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:34:03.492709Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Are They the Same? Exploring Visual Correspondence Shortcomings of Multimodal LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiangning Zhang, Lu Qi, Qianyu Zhou, Shihao Chen, Shilin Xu, Shunping Ji, Tao Zhang, Xiangtai Li, Yikang Zhou, Yunhai Tong","submitted_at":"2025-01-08T18:30:53Z","abstract_excerpt":"Recent advancements in multimodal large language models (MLLM) have shown a strong ability in visual perception, reasoning abilities, and vision-language understanding. However, the visual matching ability of MLLMs is rarely studied, despite finding the visual correspondence of objects is essential in computer vision. Our research reveals that the matching capabilities in recent MLLMs still exhibit systematic shortcomings, even with current strong MLLMs models, GPT-4o. In particular, we construct a Multimodal Visual Matching (MMVM) benchmark to fairly benchmark over 30 different MLLMs. The MMV"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.04670","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.04670/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.04670","created_at":"2026-07-05T11:34:03.492779+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.04670v3","created_at":"2026-07-05T11:34:03.492779+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.04670","created_at":"2026-07-05T11:34:03.492779+00:00"},{"alias_kind":"pith_short_12","alias_value":"CUAYLEV22OKV","created_at":"2026-07-05T11:34:03.492779+00:00"},{"alias_kind":"pith_short_16","alias_value":"CUAYLEV22OKVYAHI","created_at":"2026-07-05T11:34:03.492779+00:00"},{"alias_kind":"pith_short_8","alias_value":"CUAYLEV2","created_at":"2026-07-05T11:34:03.492779+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.22498","citing_title":"CGC: Compositional Grounded Contrast for Fine-Grained Multi-Image Understanding","ref_index":66,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CUAYLEV22OKVYAHIFW2QVJQ65B","json":"https://pith.science/pith/CUAYLEV22OKVYAHIFW2QVJQ65B.json","graph_json":"https://pith.science/api/pith-number/CUAYLEV22OKVYAHIFW2QVJQ65B/graph.json","events_json":"https://pith.science/api/pith-number/CUAYLEV22OKVYAHIFW2QVJQ65B/events.json","paper":"https://pith.science/paper/CUAYLEV2"},"agent_actions":{"view_html":"https://pith.science/pith/CUAYLEV22OKVYAHIFW2QVJQ65B","download_json":"https://pith.science/pith/CUAYLEV22OKVYAHIFW2QVJQ65B.json","view_paper":"https://pith.science/paper/CUAYLEV2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.04670&json=true","fetch_graph":"https://pith.science/api/pith-number/CUAYLEV22OKVYAHIFW2QVJQ65B/graph.json","fetch_events":"https://pith.science/api/pith-number/CUAYLEV22OKVYAHIFW2QVJQ65B/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CUAYLEV22OKVYAHIFW2QVJQ65B/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CUAYLEV22OKVYAHIFW2QVJQ65B/action/storage_attestation","attest_author":"https://pith.science/pith/CUAYLEV22OKVYAHIFW2QVJQ65B/action/author_attestation","sign_citation":"https://pith.science/pith/CUAYLEV22OKVYAHIFW2QVJQ65B/action/citation_signature","submit_replication":"https://pith.science/pith/CUAYLEV22OKVYAHIFW2QVJQ65B/action/replication_record"}},"created_at":"2026-07-05T11:34:03.492779+00:00","updated_at":"2026-07-05T11:34:03.492779+00:00"}