{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:EGNPWDTC74JVYM5GZOZRHHXP4R","short_pith_number":"pith:EGNPWDTC","schema_version":"1.0","canonical_sha256":"219afb0e62ff135c33a6cbb3139eefe4501b720e64ea4961eb9ab2307500a610","source":{"kind":"arxiv","id":"2412.20927","version":1},"attestation_state":"computed","paper":{"title":"Enhanced Multimodal RAG-LLM for Accurate Visual Question Answering","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Fei Yu, Jun Wang, Junxiao Xue, Quan Deng, Yanhao Wang, Yuehua Li","submitted_at":"2024-12-30T13:16:08Z","abstract_excerpt":"Multimodal large language models (MLLMs), such as GPT-4o, Gemini, LLaVA, and Flamingo, have made significant progress in integrating visual and textual modalities, excelling in tasks like visual question answering (VQA), image captioning, and content retrieval. They can generate coherent and contextually relevant descriptions of images. However, they still face challenges in accurately identifying and counting objects and determining their spatial locations, particularly in complex scenes with overlapping or small objects. To address these limitations, we propose a novel framework based on mul"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.20927","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-12-30T13:16:08Z","cross_cats_sorted":[],"title_canon_sha256":"ab16690a2c5a0c09f090d01a29100b5ab0876c075c19a840c28d50237e4711c0","abstract_canon_sha256":"f89a6de5d5a06a5dda0a98aa3f6b03f2fd3e3be64efbcf3b0a8c3ab22c504c91"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:55:24.907666Z","signature_b64":"D6yV9bInqdtCoZbJxaD6g8fCBfHlzviCKIgh6p72vI5PA3tLKmVh0xeBB8VylJXXUkNTdYJu8SH65azxSuwdDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"219afb0e62ff135c33a6cbb3139eefe4501b720e64ea4961eb9ab2307500a610","last_reissued_at":"2026-07-05T09:55:24.907162Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:55:24.907162Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Enhanced Multimodal RAG-LLM for Accurate Visual Question Answering","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Fei Yu, Jun Wang, Junxiao Xue, Quan Deng, Yanhao Wang, Yuehua Li","submitted_at":"2024-12-30T13:16:08Z","abstract_excerpt":"Multimodal large language models (MLLMs), such as GPT-4o, Gemini, LLaVA, and Flamingo, have made significant progress in integrating visual and textual modalities, excelling in tasks like visual question answering (VQA), image captioning, and content retrieval. They can generate coherent and contextually relevant descriptions of images. However, they still face challenges in accurately identifying and counting objects and determining their spatial locations, particularly in complex scenes with overlapping or small objects. To address these limitations, we propose a novel framework based on mul"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.20927","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.20927/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.20927","created_at":"2026-07-05T09:55:24.907221+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.20927v1","created_at":"2026-07-05T09:55:24.907221+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.20927","created_at":"2026-07-05T09:55:24.907221+00:00"},{"alias_kind":"pith_short_12","alias_value":"EGNPWDTC74JV","created_at":"2026-07-05T09:55:24.907221+00:00"},{"alias_kind":"pith_short_16","alias_value":"EGNPWDTC74JVYM5G","created_at":"2026-07-05T09:55:24.907221+00:00"},{"alias_kind":"pith_short_8","alias_value":"EGNPWDTC","created_at":"2026-07-05T09:55:24.907221+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.23145","citing_title":"UpstreamQA: A Modular Framework for Explicit Reasoning on Video Question Answering Tasks","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EGNPWDTC74JVYM5GZOZRHHXP4R","json":"https://pith.science/pith/EGNPWDTC74JVYM5GZOZRHHXP4R.json","graph_json":"https://pith.science/api/pith-number/EGNPWDTC74JVYM5GZOZRHHXP4R/graph.json","events_json":"https://pith.science/api/pith-number/EGNPWDTC74JVYM5GZOZRHHXP4R/events.json","paper":"https://pith.science/paper/EGNPWDTC"},"agent_actions":{"view_html":"https://pith.science/pith/EGNPWDTC74JVYM5GZOZRHHXP4R","download_json":"https://pith.science/pith/EGNPWDTC74JVYM5GZOZRHHXP4R.json","view_paper":"https://pith.science/paper/EGNPWDTC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.20927&json=true","fetch_graph":"https://pith.science/api/pith-number/EGNPWDTC74JVYM5GZOZRHHXP4R/graph.json","fetch_events":"https://pith.science/api/pith-number/EGNPWDTC74JVYM5GZOZRHHXP4R/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EGNPWDTC74JVYM5GZOZRHHXP4R/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EGNPWDTC74JVYM5GZOZRHHXP4R/action/storage_attestation","attest_author":"https://pith.science/pith/EGNPWDTC74JVYM5GZOZRHHXP4R/action/author_attestation","sign_citation":"https://pith.science/pith/EGNPWDTC74JVYM5GZOZRHHXP4R/action/citation_signature","submit_replication":"https://pith.science/pith/EGNPWDTC74JVYM5GZOZRHHXP4R/action/replication_record"}},"created_at":"2026-07-05T09:55:24.907221+00:00","updated_at":"2026-07-05T09:55:24.907221+00:00"}