{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:NKJXTLD4JGM2CSG2DC64KPPDFM","short_pith_number":"pith:NKJXTLD4","schema_version":"1.0","canonical_sha256":"6a9379ac7c4999a148da18bdc53de32b39da3e3c22f200c59a61fb2ee214c730","source":{"kind":"arxiv","id":"2410.08876","version":2},"attestation_state":"computed","paper":{"title":"RoRA-VLM: Robust Retrieval-Augmented Vision Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jin Di, Jingyuan Qi, Lifu Huang, Qifan Wang, Rulin Shao, Yang Chen, Yu Cheng, Zhiyang Xu","submitted_at":"2024-10-11T14:51:00Z","abstract_excerpt":"Current vision-language models (VLMs) still exhibit inferior performance on knowledge-intensive tasks, primarily due to the challenge of accurately encoding all the associations between visual objects and scenes to their corresponding entities and background knowledge. While retrieval augmentation methods offer an efficient way to integrate external knowledge, extending them to vision-language domain presents unique challenges in (1) precisely retrieving relevant information from external sources due to the inherent discrepancy within the multimodal queries, and (2) being resilient to the irre"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.08876","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-11T14:51:00Z","cross_cats_sorted":[],"title_canon_sha256":"147123ca6349e36f51708a6e105f15778d2b02951fc0962b24c908269bf564b0","abstract_canon_sha256":"157a6f39df58f07dfb5d04ca7afb88e15898c4a767b8191fd2b1c9707c0666fc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:20:25.239603Z","signature_b64":"h17C+HjgSywNPsC6I8edmMAjSxpTKc0pECVmcLplMlqRz9XZopWnZZEDOXEGv7geAtWy/UqsYufbxM4FiiIzDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6a9379ac7c4999a148da18bdc53de32b39da3e3c22f200c59a61fb2ee214c730","last_reissued_at":"2026-07-05T09:20:25.239132Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:20:25.239132Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RoRA-VLM: Robust Retrieval-Augmented Vision Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jin Di, Jingyuan Qi, Lifu Huang, Qifan Wang, Rulin Shao, Yang Chen, Yu Cheng, Zhiyang Xu","submitted_at":"2024-10-11T14:51:00Z","abstract_excerpt":"Current vision-language models (VLMs) still exhibit inferior performance on knowledge-intensive tasks, primarily due to the challenge of accurately encoding all the associations between visual objects and scenes to their corresponding entities and background knowledge. While retrieval augmentation methods offer an efficient way to integrate external knowledge, extending them to vision-language domain presents unique challenges in (1) precisely retrieving relevant information from external sources due to the inherent discrepancy within the multimodal queries, and (2) being resilient to the irre"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.08876","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.08876/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.08876","created_at":"2026-07-05T09:20:25.239187+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.08876v2","created_at":"2026-07-05T09:20:25.239187+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.08876","created_at":"2026-07-05T09:20:25.239187+00:00"},{"alias_kind":"pith_short_12","alias_value":"NKJXTLD4JGM2","created_at":"2026-07-05T09:20:25.239187+00:00"},{"alias_kind":"pith_short_16","alias_value":"NKJXTLD4JGM2CSG2","created_at":"2026-07-05T09:20:25.239187+00:00"},{"alias_kind":"pith_short_8","alias_value":"NKJXTLD4","created_at":"2026-07-05T09:20:25.239187+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07383","citing_title":"MMAgent-R$^2$: Learning to Rerank and Reject for Agentic mRAG","ref_index":21,"is_internal_anchor":true},{"citing_arxiv_id":"2606.17888","citing_title":"MathVis-Fine: Aligning Visual Supervision with Necessity via Progressive Dependency-Guided Training for Multimodal Mathematical Reasoning","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29562","citing_title":"VLA-Pro: Cross-Task Procedural Memory Transfer for Vision-Language-Action Models","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19307","citing_title":"MetaRA: Metamorphic Robustness Assessment for Multimodal Large Language Model-based Visual Question Answering Systems","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2508.05318","citing_title":"mKG-RAG: Leveraging Multimodal Knowledge Graphs in Retrieval-Augmented Generation for Knowledge-intensive VQA","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04969","citing_title":"MG$^2$-RAG: Multi-Granularity Graph for Multimodal Retrieval-Augmented Generation","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27600","citing_title":"Purifying Multimodal Retrieval: Fragment-Level Evidence Selection for RAG","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03790","citing_title":"Enhancing Visual Question Answering with Multimodal LLMs via Chain-of-Question Guided Retrieval-Augmented Generation","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07146","citing_title":"Learning to Search: A Decision-Based Agent for Knowledge-Based Visual Question Answering","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05818","citing_title":"WikiSeeker: Rethinking the Role of Vision-Language Models in Knowledge-Based Visual Question Answering","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NKJXTLD4JGM2CSG2DC64KPPDFM","json":"https://pith.science/pith/NKJXTLD4JGM2CSG2DC64KPPDFM.json","graph_json":"https://pith.science/api/pith-number/NKJXTLD4JGM2CSG2DC64KPPDFM/graph.json","events_json":"https://pith.science/api/pith-number/NKJXTLD4JGM2CSG2DC64KPPDFM/events.json","paper":"https://pith.science/paper/NKJXTLD4"},"agent_actions":{"view_html":"https://pith.science/pith/NKJXTLD4JGM2CSG2DC64KPPDFM","download_json":"https://pith.science/pith/NKJXTLD4JGM2CSG2DC64KPPDFM.json","view_paper":"https://pith.science/paper/NKJXTLD4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.08876&json=true","fetch_graph":"https://pith.science/api/pith-number/NKJXTLD4JGM2CSG2DC64KPPDFM/graph.json","fetch_events":"https://pith.science/api/pith-number/NKJXTLD4JGM2CSG2DC64KPPDFM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NKJXTLD4JGM2CSG2DC64KPPDFM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NKJXTLD4JGM2CSG2DC64KPPDFM/action/storage_attestation","attest_author":"https://pith.science/pith/NKJXTLD4JGM2CSG2DC64KPPDFM/action/author_attestation","sign_citation":"https://pith.science/pith/NKJXTLD4JGM2CSG2DC64KPPDFM/action/citation_signature","submit_replication":"https://pith.science/pith/NKJXTLD4JGM2CSG2DC64KPPDFM/action/replication_record"}},"created_at":"2026-07-05T09:20:25.239187+00:00","updated_at":"2026-07-05T09:20:25.239187+00:00"}