{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ULEQCEAKHBHAPYI5AP45IP6JVR","short_pith_number":"pith:ULEQCEAK","schema_version":"1.0","canonical_sha256":"a2c901100a384e07e11d03f9d43fc9ac5f0724811cb41e516ebfc1787bbc2454","source":{"kind":"arxiv","id":"2504.08748","version":1},"attestation_state":"computed","paper":{"title":"A Survey of Multimodal Retrieval-Augmented Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.ET","cs.LG"],"primary_cat":"cs.IR","authors_text":"Chong Chen, Lang Mei, Siyu Mo, Zhihan Yang","submitted_at":"2025-03-26T02:43:09Z","abstract_excerpt":"Multimodal Retrieval-Augmented Generation (MRAG) enhances large language models (LLMs) by integrating multimodal data (text, images, videos) into retrieval and generation processes, overcoming the limitations of text-only Retrieval-Augmented Generation (RAG). While RAG improves response accuracy by incorporating external textual knowledge, MRAG extends this framework to include multimodal retrieval and generation, leveraging contextual information from diverse data types. This approach reduces hallucinations and enhances question-answering systems by grounding responses in factual, multimodal "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.08748","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.IR","submitted_at":"2025-03-26T02:43:09Z","cross_cats_sorted":["cs.AI","cs.CL","cs.ET","cs.LG"],"title_canon_sha256":"cdd5b7a4ac0b3010d27ba11e8be39d04c95b900adc1aec49f77634019b70119c","abstract_canon_sha256":"fe7c3ac6cbb48206c7878c09cf55626be04731ac6557d6449e585596d0b54430"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:47:58.491656Z","signature_b64":"UbFiY1IXbnFEeC05Am8jFfTn1OKfBxt0bpSIFeMn0iP4bIVqODhoKq6RVH5lf7qIt/YzWQHSqIVKYxlWoBlgCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a2c901100a384e07e11d03f9d43fc9ac5f0724811cb41e516ebfc1787bbc2454","last_reissued_at":"2026-07-05T10:47:58.491138Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:47:58.491138Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Survey of Multimodal Retrieval-Augmented Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.ET","cs.LG"],"primary_cat":"cs.IR","authors_text":"Chong Chen, Lang Mei, Siyu Mo, Zhihan Yang","submitted_at":"2025-03-26T02:43:09Z","abstract_excerpt":"Multimodal Retrieval-Augmented Generation (MRAG) enhances large language models (LLMs) by integrating multimodal data (text, images, videos) into retrieval and generation processes, overcoming the limitations of text-only Retrieval-Augmented Generation (RAG). While RAG improves response accuracy by incorporating external textual knowledge, MRAG extends this framework to include multimodal retrieval and generation, leveraging contextual information from diverse data types. This approach reduces hallucinations and enhances question-answering systems by grounding responses in factual, multimodal "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.08748","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.08748/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.08748","created_at":"2026-07-05T10:47:58.491189+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.08748v1","created_at":"2026-07-05T10:47:58.491189+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.08748","created_at":"2026-07-05T10:47:58.491189+00:00"},{"alias_kind":"pith_short_12","alias_value":"ULEQCEAKHBHA","created_at":"2026-07-05T10:47:58.491189+00:00"},{"alias_kind":"pith_short_16","alias_value":"ULEQCEAKHBHAPYI5","created_at":"2026-07-05T10:47:58.491189+00:00"},{"alias_kind":"pith_short_8","alias_value":"ULEQCEAK","created_at":"2026-07-05T10:47:58.491189+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":21,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25343","citing_title":"Invoice Haystack: Benchmarking Document Retrieval and Visual Question Answering Under Strong Visual Homogeneity","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04231","citing_title":"MM-BizRAG: Rethinking Multimodal Retrieval-Augmented Generation for General Purpose Enterprise Q&A","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00685","citing_title":"M2Note: Continual Evolution of Vision Language Models via Mistake Notebook Learning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25343","citing_title":"Invoice Haystack: Benchmarking Document Retrieval and Visual Question Answering Under Strong Visual Homogeneity","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26458","citing_title":"MKG-RAG-Bench: Benchmarking Retrieval in Multimodal Knowledge Graph-Augmented Generation","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18884","citing_title":"Navigating the Emotion Tree: Hierarchical Hyperbolic RAG for Multimodal Emotion Recognition","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2505.22095","citing_title":"Mixture-of-Retrieval Experts for Reasoning-Guided Multimodal Knowledge Exploitation","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2510.15253","citing_title":"Scaling Beyond Context: A Survey of Multimodal Retrieval-Augmented Generation for Document Understanding","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2602.19549","citing_title":"Sculpting the Vector Space: Towards Efficient Multi-Vector Visual Document Retrieval via Prune-then-Merge Framework","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13277","citing_title":"Utility-Oriented Visual Evidence Selection for Multimodal Retrieval-Augmented Generation","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11864","citing_title":"Very Efficient Listwise Multimodal Reranking for Long Documents","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27600","citing_title":"Purifying Multimodal Retrieval: Fragment-Level Evidence Selection for RAG","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2601.04720","citing_title":"Qwen3-VL-Embedding and Qwen3-VL-Reranker: A Unified Framework for State-of-the-Art Multimodal Retrieval and Ranking","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08421","citing_title":"Beyond Bag-of-Patches: Learning Global Layout via Textual Supervision for Late-Interaction Visual Document Retrieval","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10253","citing_title":"Knowledge Poisoning Attacks on Medical Multi-Modal Retrieval-Augmented Generation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24564","citing_title":"MEG-RAG: Quantifying Multi-modal Evidence Grounding for Evidence Selection in RAG","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00814","citing_title":"Persistent Visual Memory: Sustaining Perception for Deep Generation in LVLMs","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12735","citing_title":"AffectAgent: Collaborative Multi-Agent Reasoning for Retrieval-Augmented Multimodal Emotion Recognition","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12890","citing_title":"Towards Long-horizon Agentic Multimodal Search","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07784","citing_title":"Automotive Engineering-Centric Agentic AI Workflow Framework","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00814","citing_title":"Persistent Visual Memory: Sustaining Perception for Deep Generation in LVLMs","ref_index":55,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ULEQCEAKHBHAPYI5AP45IP6JVR","json":"https://pith.science/pith/ULEQCEAKHBHAPYI5AP45IP6JVR.json","graph_json":"https://pith.science/api/pith-number/ULEQCEAKHBHAPYI5AP45IP6JVR/graph.json","events_json":"https://pith.science/api/pith-number/ULEQCEAKHBHAPYI5AP45IP6JVR/events.json","paper":"https://pith.science/paper/ULEQCEAK"},"agent_actions":{"view_html":"https://pith.science/pith/ULEQCEAKHBHAPYI5AP45IP6JVR","download_json":"https://pith.science/pith/ULEQCEAKHBHAPYI5AP45IP6JVR.json","view_paper":"https://pith.science/paper/ULEQCEAK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.08748&json=true","fetch_graph":"https://pith.science/api/pith-number/ULEQCEAKHBHAPYI5AP45IP6JVR/graph.json","fetch_events":"https://pith.science/api/pith-number/ULEQCEAKHBHAPYI5AP45IP6JVR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ULEQCEAKHBHAPYI5AP45IP6JVR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ULEQCEAKHBHAPYI5AP45IP6JVR/action/storage_attestation","attest_author":"https://pith.science/pith/ULEQCEAKHBHAPYI5AP45IP6JVR/action/author_attestation","sign_citation":"https://pith.science/pith/ULEQCEAKHBHAPYI5AP45IP6JVR/action/citation_signature","submit_replication":"https://pith.science/pith/ULEQCEAKHBHAPYI5AP45IP6JVR/action/replication_record"}},"created_at":"2026-07-05T10:47:58.491189+00:00","updated_at":"2026-07-05T10:47:58.491189+00:00"}