{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:Q5RHSHF2WK5RAMYP5BV4IVNIWQ","short_pith_number":"pith:Q5RHSHF2","schema_version":"1.0","canonical_sha256":"8762791cbab2bb10330fe86bc455a8b43c72b60b3cf3f68baa8b7ad2b8d169b6","source":{"kind":"arxiv","id":"2410.21220","version":1},"attestation_state":"computed","paper":{"title":"Vision Search Assistant: Empower Vision-Language Models as Multimodal Search Engines","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.IR","cs.LG"],"primary_cat":"cs.CV","authors_text":"Xiangyu Yue, Xiaohan Ding, Yiyuan Zhang, Zhixin Zhang","submitted_at":"2024-10-28T17:04:18Z","abstract_excerpt":"Search engines enable the retrieval of unknown information with texts. However, traditional methods fall short when it comes to understanding unfamiliar visual content, such as identifying an object that the model has never seen before. This challenge is particularly pronounced for large vision-language models (VLMs): if the model has not been exposed to the object depicted in an image, it struggles to generate reliable answers to the user's question regarding that image. Moreover, as new objects and events continuously emerge, frequently updating VLMs is impractical due to heavy computational"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.21220","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-10-28T17:04:18Z","cross_cats_sorted":["cs.AI","cs.IR","cs.LG"],"title_canon_sha256":"d5bb16243869c7c29f22dc6937be4cd4a9147fab3f957c16e8544ed8703b3a27","abstract_canon_sha256":"303132fe9115b01b20f5f7c49c39a891aff6cf424928e730dc00d375e6d5db91"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:27:17.066535Z","signature_b64":"2DN78cr8I7hwdDfhWbbq31tzqICZq9x2TGSdQTYfqzoWViu4oBDeKLquAIeOexbf4lSRAl8qRrkuszC0JerWDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8762791cbab2bb10330fe86bc455a8b43c72b60b3cf3f68baa8b7ad2b8d169b6","last_reissued_at":"2026-07-05T09:27:17.065896Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:27:17.065896Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Vision Search Assistant: Empower Vision-Language Models as Multimodal Search Engines","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.IR","cs.LG"],"primary_cat":"cs.CV","authors_text":"Xiangyu Yue, Xiaohan Ding, Yiyuan Zhang, Zhixin Zhang","submitted_at":"2024-10-28T17:04:18Z","abstract_excerpt":"Search engines enable the retrieval of unknown information with texts. However, traditional methods fall short when it comes to understanding unfamiliar visual content, such as identifying an object that the model has never seen before. This challenge is particularly pronounced for large vision-language models (VLMs): if the model has not been exposed to the object depicted in an image, it struggles to generate reliable answers to the user's question regarding that image. Moreover, as new objects and events continuously emerge, frequently updating VLMs is impractical due to heavy computational"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.21220","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.21220/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.21220","created_at":"2026-07-05T09:27:17.065972+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.21220v1","created_at":"2026-07-05T09:27:17.065972+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.21220","created_at":"2026-07-05T09:27:17.065972+00:00"},{"alias_kind":"pith_short_12","alias_value":"Q5RHSHF2WK5R","created_at":"2026-07-05T09:27:17.065972+00:00"},{"alias_kind":"pith_short_16","alias_value":"Q5RHSHF2WK5RAMYP","created_at":"2026-07-05T09:27:17.065972+00:00"},{"alias_kind":"pith_short_8","alias_value":"Q5RHSHF2","created_at":"2026-07-05T09:27:17.065972+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2506.20670","citing_title":"MMSearch-R1: Incentivizing LMMs to Search","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20486","citing_title":"ProMMSearchAgent: A Generalizable Multimodal Search Agent Trained with Process-Oriented Rewards","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19264","citing_title":"DR-MMSearchAgent: Deepening Reasoning in Multimodal Search Agents","ref_index":65,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Q5RHSHF2WK5RAMYP5BV4IVNIWQ","json":"https://pith.science/pith/Q5RHSHF2WK5RAMYP5BV4IVNIWQ.json","graph_json":"https://pith.science/api/pith-number/Q5RHSHF2WK5RAMYP5BV4IVNIWQ/graph.json","events_json":"https://pith.science/api/pith-number/Q5RHSHF2WK5RAMYP5BV4IVNIWQ/events.json","paper":"https://pith.science/paper/Q5RHSHF2"},"agent_actions":{"view_html":"https://pith.science/pith/Q5RHSHF2WK5RAMYP5BV4IVNIWQ","download_json":"https://pith.science/pith/Q5RHSHF2WK5RAMYP5BV4IVNIWQ.json","view_paper":"https://pith.science/paper/Q5RHSHF2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.21220&json=true","fetch_graph":"https://pith.science/api/pith-number/Q5RHSHF2WK5RAMYP5BV4IVNIWQ/graph.json","fetch_events":"https://pith.science/api/pith-number/Q5RHSHF2WK5RAMYP5BV4IVNIWQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Q5RHSHF2WK5RAMYP5BV4IVNIWQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Q5RHSHF2WK5RAMYP5BV4IVNIWQ/action/storage_attestation","attest_author":"https://pith.science/pith/Q5RHSHF2WK5RAMYP5BV4IVNIWQ/action/author_attestation","sign_citation":"https://pith.science/pith/Q5RHSHF2WK5RAMYP5BV4IVNIWQ/action/citation_signature","submit_replication":"https://pith.science/pith/Q5RHSHF2WK5RAMYP5BV4IVNIWQ/action/replication_record"}},"created_at":"2026-07-05T09:27:17.065972+00:00","updated_at":"2026-07-05T09:27:17.065972+00:00"}