{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RE3HGEQULCQ4MIT6WLHKQDOQOW","short_pith_number":"pith:RE3HGEQU","schema_version":"1.0","canonical_sha256":"893673121458a1c6227eb2cea80dd0758fbba53f04f1740b18c499653d77797d","source":{"kind":"arxiv","id":"2402.12728","version":2},"attestation_state":"computed","paper":{"title":"Modality-Aware Integration with Large Language Models for Knowledge-based Visual Question Answering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.IR","cs.LG"],"primary_cat":"cs.CV","authors_text":"Daochen Zha, Huachi Zhou, Junnan Dong, Pai Zheng, Qinggang Zhang, Xiao Huang","submitted_at":"2024-02-20T05:32:24Z","abstract_excerpt":"Knowledge-based visual question answering (KVQA) has been extensively studied to answer visual questions with external knowledge, e.g., knowledge graphs (KGs). While several attempts have been proposed to leverage large language models (LLMs) as an implicit knowledge source, it remains challenging since LLMs may generate hallucinations. Moreover, multiple knowledge sources, e.g., images, KGs and LLMs, cannot be readily aligned for complex scenarios. To tackle these, we present a novel modality-aware integration with LLMs for KVQA (MAIL). It carefully leverages multimodal knowledge for both ima"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.12728","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-02-20T05:32:24Z","cross_cats_sorted":["cs.AI","cs.CL","cs.IR","cs.LG"],"title_canon_sha256":"2b90c41a30c14f1a4997fdc8e52e93f0125cc34228d63b0492329e353626c6d2","abstract_canon_sha256":"9e9849ae07a0f6cba63d9f4e7cf830df143fb697c26d884c57393ad90c8aeab0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:51:39.424019Z","signature_b64":"rzsIYQahM8wDsV+w5R48Ya2lxUu6/RpQ5+rNrNUnvZzsF0AU4BGZ2ny18UieIAvLnVLa3m68HWAHuD/4b9XxDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"893673121458a1c6227eb2cea80dd0758fbba53f04f1740b18c499653d77797d","last_reissued_at":"2026-07-05T07:51:39.423486Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:51:39.423486Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Modality-Aware Integration with Large Language Models for Knowledge-based Visual Question Answering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.IR","cs.LG"],"primary_cat":"cs.CV","authors_text":"Daochen Zha, Huachi Zhou, Junnan Dong, Pai Zheng, Qinggang Zhang, Xiao Huang","submitted_at":"2024-02-20T05:32:24Z","abstract_excerpt":"Knowledge-based visual question answering (KVQA) has been extensively studied to answer visual questions with external knowledge, e.g., knowledge graphs (KGs). While several attempts have been proposed to leverage large language models (LLMs) as an implicit knowledge source, it remains challenging since LLMs may generate hallucinations. Moreover, multiple knowledge sources, e.g., images, KGs and LLMs, cannot be readily aligned for complex scenarios. To tackle these, we present a novel modality-aware integration with LLMs for KVQA (MAIL). It carefully leverages multimodal knowledge for both ima"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.12728","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.12728/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.12728","created_at":"2026-07-05T07:51:39.423550+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.12728v2","created_at":"2026-07-05T07:51:39.423550+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.12728","created_at":"2026-07-05T07:51:39.423550+00:00"},{"alias_kind":"pith_short_12","alias_value":"RE3HGEQULCQ4","created_at":"2026-07-05T07:51:39.423550+00:00"},{"alias_kind":"pith_short_16","alias_value":"RE3HGEQULCQ4MIT6","created_at":"2026-07-05T07:51:39.423550+00:00"},{"alias_kind":"pith_short_8","alias_value":"RE3HGEQU","created_at":"2026-07-05T07:51:39.423550+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.31010","citing_title":"MoG: Mixture of Experts for Graph-based Retrieval-Augmented Generation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17458","citing_title":"EHRAG: Bridging Semantic Gaps in Lightweight GraphRAG via Hybrid Hypergraph Construction and Retrieval","ref_index":259,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RE3HGEQULCQ4MIT6WLHKQDOQOW","json":"https://pith.science/pith/RE3HGEQULCQ4MIT6WLHKQDOQOW.json","graph_json":"https://pith.science/api/pith-number/RE3HGEQULCQ4MIT6WLHKQDOQOW/graph.json","events_json":"https://pith.science/api/pith-number/RE3HGEQULCQ4MIT6WLHKQDOQOW/events.json","paper":"https://pith.science/paper/RE3HGEQU"},"agent_actions":{"view_html":"https://pith.science/pith/RE3HGEQULCQ4MIT6WLHKQDOQOW","download_json":"https://pith.science/pith/RE3HGEQULCQ4MIT6WLHKQDOQOW.json","view_paper":"https://pith.science/paper/RE3HGEQU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.12728&json=true","fetch_graph":"https://pith.science/api/pith-number/RE3HGEQULCQ4MIT6WLHKQDOQOW/graph.json","fetch_events":"https://pith.science/api/pith-number/RE3HGEQULCQ4MIT6WLHKQDOQOW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RE3HGEQULCQ4MIT6WLHKQDOQOW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RE3HGEQULCQ4MIT6WLHKQDOQOW/action/storage_attestation","attest_author":"https://pith.science/pith/RE3HGEQULCQ4MIT6WLHKQDOQOW/action/author_attestation","sign_citation":"https://pith.science/pith/RE3HGEQULCQ4MIT6WLHKQDOQOW/action/citation_signature","submit_replication":"https://pith.science/pith/RE3HGEQULCQ4MIT6WLHKQDOQOW/action/replication_record"}},"created_at":"2026-07-05T07:51:39.423550+00:00","updated_at":"2026-07-05T07:51:39.423550+00:00"}