{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:SJP4TDVZZOV4H6T4HZ45NTFWAV","short_pith_number":"pith:SJP4TDVZ","schema_version":"1.0","canonical_sha256":"925fc98eb9cbabc3fa7c3e79d6ccb6054f4a5882a6812e4b3f6a0f0e596560b9","source":{"kind":"arxiv","id":"2502.01549","version":1},"attestation_state":"computed","paper":{"title":"VideoRAG: Retrieval-Augmented Generation with Extreme Long-Context Videos","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.IR","authors_text":"Chao Huang, Dawei Yin, Lingrui Xu, Long Xia, Shuaiqiang Wang, Xubin Ren","submitted_at":"2025-02-03T17:30:19Z","abstract_excerpt":"Retrieval-Augmented Generation (RAG) has demonstrated remarkable success in enhancing Large Language Models (LLMs) through external knowledge integration, yet its application has primarily focused on textual content, leaving the rich domain of multi-modal video knowledge predominantly unexplored. This paper introduces VideoRAG, the first retrieval-augmented generation framework specifically designed for processing and understanding extremely long-context videos. Our core innovation lies in its dual-channel architecture that seamlessly integrates (i) graph-based textual knowledge grounding for "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.01549","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.IR","submitted_at":"2025-02-03T17:30:19Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"9ba23bbb1362d67ee701b142f3d3ae1bae03d8d283f5e6abc3b4ff3f807ee4ac","abstract_canon_sha256":"1057de4c39dcffcdef976cc9e29053e681c6439f12ffede4847d751042dc0cce"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:08:56.567758Z","signature_b64":"rpdORDnXRJRXQ9b6Tu9ZfwVLsFgQKc41m2T7XAX+FRr5v0EaXhrjuk5smnuSPKrTl9IONqq6DdX5qVG9EXg6DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"925fc98eb9cbabc3fa7c3e79d6ccb6054f4a5882a6812e4b3f6a0f0e596560b9","last_reissued_at":"2026-07-05T10:08:56.567271Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:08:56.567271Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VideoRAG: Retrieval-Augmented Generation with Extreme Long-Context Videos","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.IR","authors_text":"Chao Huang, Dawei Yin, Lingrui Xu, Long Xia, Shuaiqiang Wang, Xubin Ren","submitted_at":"2025-02-03T17:30:19Z","abstract_excerpt":"Retrieval-Augmented Generation (RAG) has demonstrated remarkable success in enhancing Large Language Models (LLMs) through external knowledge integration, yet its application has primarily focused on textual content, leaving the rich domain of multi-modal video knowledge predominantly unexplored. This paper introduces VideoRAG, the first retrieval-augmented generation framework specifically designed for processing and understanding extremely long-context videos. Our core innovation lies in its dual-channel architecture that seamlessly integrates (i) graph-based textual knowledge grounding for "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.01549","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.01549/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.01549","created_at":"2026-07-05T10:08:56.567342+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.01549v1","created_at":"2026-07-05T10:08:56.567342+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.01549","created_at":"2026-07-05T10:08:56.567342+00:00"},{"alias_kind":"pith_short_12","alias_value":"SJP4TDVZZOV4","created_at":"2026-07-05T10:08:56.567342+00:00"},{"alias_kind":"pith_short_16","alias_value":"SJP4TDVZZOV4H6T4","created_at":"2026-07-05T10:08:56.567342+00:00"},{"alias_kind":"pith_short_8","alias_value":"SJP4TDVZ","created_at":"2026-07-05T10:08:56.567342+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24187","citing_title":"Towards Fast and Effective Long Video Understanding of Multimodal Large Language Models via Adaptive Quasi-Gaussian Sampling","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07433","citing_title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","ref_index":208,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00446","citing_title":"VideoSearch-R1: Iterative Video Retrieval and Reasoning via Soft Query Refinement","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00825","citing_title":"SuperMemory-VQA: An Egocentric Visual Question-Answering Benchmark for Long-Horizon Memory","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2512.20136","citing_title":"M$^3$KG-RAG: Multi-hop Multimodal Knowledge Graph-enhanced Retrieval-Augmented Generation","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2602.20913","citing_title":"LongVideo-R1: Smart Navigation for Low-cost Long Video Understanding","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03264","citing_title":"SafeScreen: A Safety-First Screening Framework for Personalized Video Retrieval for Vulnerable Users","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23145","citing_title":"UpstreamQA: A Modular Framework for Explicit Reasoning on Video Question Answering Tasks","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06185","citing_title":"Event-Causal RAG: A Retrieval-Augmented Generation Framework for Long Video Reasoning in Complex Scenarios","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05418","citing_title":"VideoStir: Understanding Long Videos via Spatio-Temporally Structured and Intent-Aware RAG","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06156","citing_title":"MMEmb-R1: Reasoning-Enhanced Multimodal Embedding with Pair-Aware Selection and Adaptive Control","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04901","citing_title":"FileGram: Grounding Agent Personalization in File-System Behavioral Traces","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04875","citing_title":"DIRECT: Video Mashup Creation via Hierarchical Multi-Agent Planning and Intent-Guided Editing","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04372","citing_title":"Graph-to-Frame RAG: Visual-Space Knowledge Fusion for Training-Free and Auditable Video Reasoning","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05079","citing_title":"SVAgent: Storyline-Guided Long Video Understanding via Cross-Modal Multi-Agent Collaboration","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SJP4TDVZZOV4H6T4HZ45NTFWAV","json":"https://pith.science/pith/SJP4TDVZZOV4H6T4HZ45NTFWAV.json","graph_json":"https://pith.science/api/pith-number/SJP4TDVZZOV4H6T4HZ45NTFWAV/graph.json","events_json":"https://pith.science/api/pith-number/SJP4TDVZZOV4H6T4HZ45NTFWAV/events.json","paper":"https://pith.science/paper/SJP4TDVZ"},"agent_actions":{"view_html":"https://pith.science/pith/SJP4TDVZZOV4H6T4HZ45NTFWAV","download_json":"https://pith.science/pith/SJP4TDVZZOV4H6T4HZ45NTFWAV.json","view_paper":"https://pith.science/paper/SJP4TDVZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.01549&json=true","fetch_graph":"https://pith.science/api/pith-number/SJP4TDVZZOV4H6T4HZ45NTFWAV/graph.json","fetch_events":"https://pith.science/api/pith-number/SJP4TDVZZOV4H6T4HZ45NTFWAV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SJP4TDVZZOV4H6T4HZ45NTFWAV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SJP4TDVZZOV4H6T4HZ45NTFWAV/action/storage_attestation","attest_author":"https://pith.science/pith/SJP4TDVZZOV4H6T4HZ45NTFWAV/action/author_attestation","sign_citation":"https://pith.science/pith/SJP4TDVZZOV4H6T4HZ45NTFWAV/action/citation_signature","submit_replication":"https://pith.science/pith/SJP4TDVZZOV4H6T4HZ45NTFWAV/action/replication_record"}},"created_at":"2026-07-05T10:08:56.567342+00:00","updated_at":"2026-07-05T10:08:56.567342+00:00"}