{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:Z6NXDNTUAMTSF2JYMRLTT6VPVE","short_pith_number":"pith:Z6NXDNTU","schema_version":"1.0","canonical_sha256":"cf9b71b674032722e938645739faafa936b600b7100f40cff9c1d89e99f53ada","source":{"kind":"arxiv","id":"2503.18016","version":1},"attestation_state":"computed","paper":{"title":"Retrieval Augmented Generation and Understanding in Vision: A Survey and New Outlook","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bin Ren, Danda Paudel, Haiwei Xue, Luc Van Gool, Lutao Jiang, Nicu Sebe, Xuming Hu, Xu Zheng, Yuanhuiyi Lyu, Ziqiao Weng","submitted_at":"2025-03-23T10:33:28Z","abstract_excerpt":"Retrieval-augmented generation (RAG) has emerged as a pivotal technique in artificial intelligence (AI), particularly in enhancing the capabilities of large language models (LLMs) by enabling access to external, reliable, and up-to-date knowledge sources. In the context of AI-Generated Content (AIGC), RAG has proven invaluable by augmenting model outputs with supplementary, relevant information, thus improving their quality. Recently, the potential of RAG has extended beyond natural language processing, with emerging methods integrating retrieval-augmented strategies into the computer vision ("},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.18016","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-03-23T10:33:28Z","cross_cats_sorted":[],"title_canon_sha256":"6499b5fad055533f2cd5d3ac97965eb8405546f705342961173f790124f101db","abstract_canon_sha256":"d8088690cfd84d2b302168e02162194b76a74ef2b216bae16c1b88239ca7c554"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:38:00.553227Z","signature_b64":"Wjca1CVdJ1q2B5Af8FV5cCedqSD07UqXAUBbOt5s1sdgu6ebl1f7vbqUaXdCKGKBzEF97ziFkjhTTpOlDkTZCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cf9b71b674032722e938645739faafa936b600b7100f40cff9c1d89e99f53ada","last_reissued_at":"2026-07-05T10:38:00.552259Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:38:00.552259Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Retrieval Augmented Generation and Understanding in Vision: A Survey and New Outlook","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bin Ren, Danda Paudel, Haiwei Xue, Luc Van Gool, Lutao Jiang, Nicu Sebe, Xuming Hu, Xu Zheng, Yuanhuiyi Lyu, Ziqiao Weng","submitted_at":"2025-03-23T10:33:28Z","abstract_excerpt":"Retrieval-augmented generation (RAG) has emerged as a pivotal technique in artificial intelligence (AI), particularly in enhancing the capabilities of large language models (LLMs) by enabling access to external, reliable, and up-to-date knowledge sources. In the context of AI-Generated Content (AIGC), RAG has proven invaluable by augmenting model outputs with supplementary, relevant information, thus improving their quality. Recently, the potential of RAG has extended beyond natural language processing, with emerging methods integrating retrieval-augmented strategies into the computer vision ("},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.18016","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.18016/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.18016","created_at":"2026-07-05T10:38:00.552410+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.18016v1","created_at":"2026-07-05T10:38:00.552410+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.18016","created_at":"2026-07-05T10:38:00.552410+00:00"},{"alias_kind":"pith_short_12","alias_value":"Z6NXDNTUAMTS","created_at":"2026-07-05T10:38:00.552410+00:00"},{"alias_kind":"pith_short_16","alias_value":"Z6NXDNTUAMTSF2JY","created_at":"2026-07-05T10:38:00.552410+00:00"},{"alias_kind":"pith_short_8","alias_value":"Z6NXDNTU","created_at":"2026-07-05T10:38:00.552410+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05927","citing_title":"CMDR: Contextual Multimodal Document Retrieval","ref_index":65,"is_internal_anchor":true},{"citing_arxiv_id":"2601.21262","citing_title":"CausalEmbed: Auto-Regressive Multi-Vector Generation in Latent Space for Visual Document Embedding","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14192","citing_title":"Why Retrieval-Augmented Generation Fails: A Graph Perspective","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12335","citing_title":"EHR-RAGp: Retrieval-Augmented Prototype-Guided Foundation Model for Electronic Health Records","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24469","citing_title":"Geometric Analysis of Self-Supervised Vision Representations for Semantic Image Retrieval","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17889","citing_title":"AeroRAG: Structured Multimodal Retrieval-Augmented LLM for Fine-Grained Aerial Visual Reasoning","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Z6NXDNTUAMTSF2JYMRLTT6VPVE","json":"https://pith.science/pith/Z6NXDNTUAMTSF2JYMRLTT6VPVE.json","graph_json":"https://pith.science/api/pith-number/Z6NXDNTUAMTSF2JYMRLTT6VPVE/graph.json","events_json":"https://pith.science/api/pith-number/Z6NXDNTUAMTSF2JYMRLTT6VPVE/events.json","paper":"https://pith.science/paper/Z6NXDNTU"},"agent_actions":{"view_html":"https://pith.science/pith/Z6NXDNTUAMTSF2JYMRLTT6VPVE","download_json":"https://pith.science/pith/Z6NXDNTUAMTSF2JYMRLTT6VPVE.json","view_paper":"https://pith.science/paper/Z6NXDNTU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.18016&json=true","fetch_graph":"https://pith.science/api/pith-number/Z6NXDNTUAMTSF2JYMRLTT6VPVE/graph.json","fetch_events":"https://pith.science/api/pith-number/Z6NXDNTUAMTSF2JYMRLTT6VPVE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Z6NXDNTUAMTSF2JYMRLTT6VPVE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Z6NXDNTUAMTSF2JYMRLTT6VPVE/action/storage_attestation","attest_author":"https://pith.science/pith/Z6NXDNTUAMTSF2JYMRLTT6VPVE/action/author_attestation","sign_citation":"https://pith.science/pith/Z6NXDNTUAMTSF2JYMRLTT6VPVE/action/citation_signature","submit_replication":"https://pith.science/pith/Z6NXDNTUAMTSF2JYMRLTT6VPVE/action/replication_record"}},"created_at":"2026-07-05T10:38:00.552410+00:00","updated_at":"2026-07-05T10:38:00.552410+00:00"}