{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OMAI2IBW7UDLEHCHXAR4KH6P4Q","short_pith_number":"pith:OMAI2IBW","schema_version":"1.0","canonical_sha256":"73008d2036fd06b21c47b823c51fcfe43c424040929b4e6633af8a3fd36e831a","source":{"kind":"arxiv","id":"2409.14083","version":1},"attestation_state":"computed","paper":{"title":"SURf: Teaching Large Vision-Language Models to Selectively Utilize Retrieved Information","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiashuo Sun, Jihai Zhang, Xiaoye Qu, Yu Cheng, Yucheng Zhou, Zhaochen Su","submitted_at":"2024-09-21T09:36:14Z","abstract_excerpt":"Large Vision-Language Models (LVLMs) have become pivotal at the intersection of computer vision and natural language processing. However, the full potential of LVLMs Retrieval-Augmented Generation (RAG) capabilities remains underutilized. Existing works either focus solely on the text modality or are limited to specific tasks. Moreover, most LVLMs struggle to selectively utilize retrieved information and are sensitive to irrelevant or misleading references. To address these challenges, we propose a self-refinement framework designed to teach LVLMs to Selectively Utilize Retrieved Information ("},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.14083","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-09-21T09:36:14Z","cross_cats_sorted":[],"title_canon_sha256":"541773c28e8dfd316ee0daf1e084c29ca79222340e73d8e8c5f52f8ab710cc15","abstract_canon_sha256":"e125dc76a3681333d4a711a2998dab5ff92f90914cdaa13dcb190eae6bce6ace"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:10:11.395057Z","signature_b64":"SNT6BK2BTDtToYQ4DsQ7Fskn8qcugykBydWZLmqiOltOt7CsHdrdw3UhODSMt+blTL7g6T2wcC3I8E13gt9hDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"73008d2036fd06b21c47b823c51fcfe43c424040929b4e6633af8a3fd36e831a","last_reissued_at":"2026-07-05T09:10:11.394720Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:10:11.394720Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SURf: Teaching Large Vision-Language Models to Selectively Utilize Retrieved Information","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiashuo Sun, Jihai Zhang, Xiaoye Qu, Yu Cheng, Yucheng Zhou, Zhaochen Su","submitted_at":"2024-09-21T09:36:14Z","abstract_excerpt":"Large Vision-Language Models (LVLMs) have become pivotal at the intersection of computer vision and natural language processing. However, the full potential of LVLMs Retrieval-Augmented Generation (RAG) capabilities remains underutilized. Existing works either focus solely on the text modality or are limited to specific tasks. Moreover, most LVLMs struggle to selectively utilize retrieved information and are sensitive to irrelevant or misleading references. To address these challenges, we propose a self-refinement framework designed to teach LVLMs to Selectively Utilize Retrieved Information ("},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.14083","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.14083/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.14083","created_at":"2026-07-05T09:10:11.394776+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.14083v1","created_at":"2026-07-05T09:10:11.394776+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.14083","created_at":"2026-07-05T09:10:11.394776+00:00"},{"alias_kind":"pith_short_12","alias_value":"OMAI2IBW7UDL","created_at":"2026-07-05T09:10:11.394776+00:00"},{"alias_kind":"pith_short_16","alias_value":"OMAI2IBW7UDLEHCH","created_at":"2026-07-05T09:10:11.394776+00:00"},{"alias_kind":"pith_short_8","alias_value":"OMAI2IBW","created_at":"2026-07-05T09:10:11.394776+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.09787","citing_title":"A Multimodal Multi-Agent Framework for Radiology Report Generation","ref_index":40,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OMAI2IBW7UDLEHCHXAR4KH6P4Q","json":"https://pith.science/pith/OMAI2IBW7UDLEHCHXAR4KH6P4Q.json","graph_json":"https://pith.science/api/pith-number/OMAI2IBW7UDLEHCHXAR4KH6P4Q/graph.json","events_json":"https://pith.science/api/pith-number/OMAI2IBW7UDLEHCHXAR4KH6P4Q/events.json","paper":"https://pith.science/paper/OMAI2IBW"},"agent_actions":{"view_html":"https://pith.science/pith/OMAI2IBW7UDLEHCHXAR4KH6P4Q","download_json":"https://pith.science/pith/OMAI2IBW7UDLEHCHXAR4KH6P4Q.json","view_paper":"https://pith.science/paper/OMAI2IBW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.14083&json=true","fetch_graph":"https://pith.science/api/pith-number/OMAI2IBW7UDLEHCHXAR4KH6P4Q/graph.json","fetch_events":"https://pith.science/api/pith-number/OMAI2IBW7UDLEHCHXAR4KH6P4Q/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OMAI2IBW7UDLEHCHXAR4KH6P4Q/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OMAI2IBW7UDLEHCHXAR4KH6P4Q/action/storage_attestation","attest_author":"https://pith.science/pith/OMAI2IBW7UDLEHCHXAR4KH6P4Q/action/author_attestation","sign_citation":"https://pith.science/pith/OMAI2IBW7UDLEHCHXAR4KH6P4Q/action/citation_signature","submit_replication":"https://pith.science/pith/OMAI2IBW7UDLEHCHXAR4KH6P4Q/action/replication_record"}},"created_at":"2026-07-05T09:10:11.394776+00:00","updated_at":"2026-07-05T09:10:11.394776+00:00"}