{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:2J2HIBX2VNC3ZWA53X42ZOYLUO","short_pith_number":"pith:2J2HIBX2","schema_version":"1.0","canonical_sha256":"d2747406faab45bcd81dddf9acbb0ba3831c560a76cf6cc068f262b4ba92302e","source":{"kind":"arxiv","id":"2504.10074","version":3},"attestation_state":"computed","paper":{"title":"MMKB-RAG: A Multi-Modal Knowledge-Based Retrieval-Augmented Generation Framework","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Bo Zheng, Jinsong Lan, Shuai Xiao, Xiaoyong Zhu, Yi An, Yixuan Huang, Zhiyao Guo, Zihan Ling","submitted_at":"2025-04-14T10:19:47Z","abstract_excerpt":"Recent advancements in large language models (LLMs) and multi-modal LLMs have been remarkable. However, these models still rely solely on their parametric knowledge, which limits their ability to generate up-to-date information and increases the risk of producing erroneous content. Retrieval-Augmented Generation (RAG) partially mitigates these challenges by incorporating external data sources, yet the reliance on databases and retrieval systems can introduce irrelevant or inaccurate documents, ultimately undermining both performance and reasoning quality. In this paper, we propose Multi-Modal "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.10074","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-04-14T10:19:47Z","cross_cats_sorted":[],"title_canon_sha256":"7557a9de1a82da54371a6d4d58a029b18058a0b404e101a0b8dded055e9683f0","abstract_canon_sha256":"b7d1a3caeed8262d3992cc6d4a4dcad482bf9f8bfe1fe2235eb3700438095ce7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:51:36.699985Z","signature_b64":"u/iyd3uo5kp3M+FJ5swz83UmwPFRnUlpgCq1c9bCqIf4eXVJGQPgdn2Q1O3xTu50FJq48kIFZwLPbmcuh473DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d2747406faab45bcd81dddf9acbb0ba3831c560a76cf6cc068f262b4ba92302e","last_reissued_at":"2026-07-05T10:51:36.699472Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:51:36.699472Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MMKB-RAG: A Multi-Modal Knowledge-Based Retrieval-Augmented Generation Framework","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Bo Zheng, Jinsong Lan, Shuai Xiao, Xiaoyong Zhu, Yi An, Yixuan Huang, Zhiyao Guo, Zihan Ling","submitted_at":"2025-04-14T10:19:47Z","abstract_excerpt":"Recent advancements in large language models (LLMs) and multi-modal LLMs have been remarkable. However, these models still rely solely on their parametric knowledge, which limits their ability to generate up-to-date information and increases the risk of producing erroneous content. Retrieval-Augmented Generation (RAG) partially mitigates these challenges by incorporating external data sources, yet the reliance on databases and retrieval systems can introduce irrelevant or inaccurate documents, ultimately undermining both performance and reasoning quality. In this paper, we propose Multi-Modal "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.10074","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.10074/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.10074","created_at":"2026-07-05T10:51:36.699536+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.10074v3","created_at":"2026-07-05T10:51:36.699536+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.10074","created_at":"2026-07-05T10:51:36.699536+00:00"},{"alias_kind":"pith_short_12","alias_value":"2J2HIBX2VNC3","created_at":"2026-07-05T10:51:36.699536+00:00"},{"alias_kind":"pith_short_16","alias_value":"2J2HIBX2VNC3ZWA5","created_at":"2026-07-05T10:51:36.699536+00:00"},{"alias_kind":"pith_short_8","alias_value":"2J2HIBX2","created_at":"2026-07-05T10:51:36.699536+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07383","citing_title":"MMAgent-R$^2$: Learning to Rerank and Reject for Agentic mRAG","ref_index":18,"is_internal_anchor":true},{"citing_arxiv_id":"2606.27974","citing_title":"ProMSA:Progressive Multimodal Search Agents for Knowledge-Based Visual Question Answering","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19307","citing_title":"MetaRA: Metamorphic Robustness Assessment for Multimodal Large Language Model-based Visual Question Answering Systems","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2509.00798","citing_title":"Progressive Multimodal Search and Reasoning for Knowledge-Intensive Visual Question Answering","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2601.13856","citing_title":"QKVQA: Question-Focused Filtering for Knowledge-based VQA","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2602.00104","citing_title":"R3G: A Reasoning-Retrieval-Reranking Framework for Vision-Centric Answer Generation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03790","citing_title":"Enhancing Visual Question Answering with Multimodal LLMs via Chain-of-Question Guided Retrieval-Augmented Generation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05268","citing_title":"Region-R1: Reinforcing Query-Side Region Cropping for Multi-Modal Re-Ranking","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2J2HIBX2VNC3ZWA53X42ZOYLUO","json":"https://pith.science/pith/2J2HIBX2VNC3ZWA53X42ZOYLUO.json","graph_json":"https://pith.science/api/pith-number/2J2HIBX2VNC3ZWA53X42ZOYLUO/graph.json","events_json":"https://pith.science/api/pith-number/2J2HIBX2VNC3ZWA53X42ZOYLUO/events.json","paper":"https://pith.science/paper/2J2HIBX2"},"agent_actions":{"view_html":"https://pith.science/pith/2J2HIBX2VNC3ZWA53X42ZOYLUO","download_json":"https://pith.science/pith/2J2HIBX2VNC3ZWA53X42ZOYLUO.json","view_paper":"https://pith.science/paper/2J2HIBX2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.10074&json=true","fetch_graph":"https://pith.science/api/pith-number/2J2HIBX2VNC3ZWA53X42ZOYLUO/graph.json","fetch_events":"https://pith.science/api/pith-number/2J2HIBX2VNC3ZWA53X42ZOYLUO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2J2HIBX2VNC3ZWA53X42ZOYLUO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2J2HIBX2VNC3ZWA53X42ZOYLUO/action/storage_attestation","attest_author":"https://pith.science/pith/2J2HIBX2VNC3ZWA53X42ZOYLUO/action/author_attestation","sign_citation":"https://pith.science/pith/2J2HIBX2VNC3ZWA53X42ZOYLUO/action/citation_signature","submit_replication":"https://pith.science/pith/2J2HIBX2VNC3ZWA53X42ZOYLUO/action/replication_record"}},"created_at":"2026-07-05T10:51:36.699536+00:00","updated_at":"2026-07-05T10:51:36.699536+00:00"}