{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:6CKKRTK7MMMIBDH5XRBNAREW74","short_pith_number":"pith:6CKKRTK7","schema_version":"1.0","canonical_sha256":"f094a8cd5f6318808cfdbc42d04496ff09fd2fc23954ce789398fa81da0feda0","source":{"kind":"arxiv","id":"2508.01546","version":1},"attestation_state":"computed","paper":{"title":"E-VRAG: Enhancing Long Video Understanding with Resource-Efficient Retrieval Augmented Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Junkang Zhang, Qiang Wang, Yi Liu, Zeyu Xu","submitted_at":"2025-08-03T02:09:54Z","abstract_excerpt":"Vision-Language Models (VLMs) have enabled substantial progress in video understanding by leveraging cross-modal reasoning capabilities. However, their effectiveness is limited by the restricted context window and the high computational cost required to process long videos with thousands of frames. Retrieval-augmented generation (RAG) addresses this challenge by selecting only the most relevant frames as input, thereby reducing the computational burden. Nevertheless, existing video RAG methods struggle to balance retrieval efficiency and accuracy, particularly when handling diverse and complex"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.01546","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-08-03T02:09:54Z","cross_cats_sorted":[],"title_canon_sha256":"94b925650bebd05feddce18c6be36f93e540069fa973cd7c0f08508b9b31f054","abstract_canon_sha256":"ae297b8f7aa19295d3480fa72d302af42279e9c18346011a3f49c09b746af5b7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:47:40.888425Z","signature_b64":"d2/RWEk1rvjIGhy0lbBr5s6vRNHsOBzwuk91UkMPXZcEZnWAIzY6HIp7q9SEts4LWbdS1/As0z8tQaBaNre1Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f094a8cd5f6318808cfdbc42d04496ff09fd2fc23954ce789398fa81da0feda0","last_reissued_at":"2026-07-05T11:47:40.887974Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:47:40.887974Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"E-VRAG: Enhancing Long Video Understanding with Resource-Efficient Retrieval Augmented Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Junkang Zhang, Qiang Wang, Yi Liu, Zeyu Xu","submitted_at":"2025-08-03T02:09:54Z","abstract_excerpt":"Vision-Language Models (VLMs) have enabled substantial progress in video understanding by leveraging cross-modal reasoning capabilities. However, their effectiveness is limited by the restricted context window and the high computational cost required to process long videos with thousands of frames. Retrieval-augmented generation (RAG) addresses this challenge by selecting only the most relevant frames as input, thereby reducing the computational burden. Nevertheless, existing video RAG methods struggle to balance retrieval efficiency and accuracy, particularly when handling diverse and complex"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.01546","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.01546/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.01546","created_at":"2026-07-05T11:47:40.888034+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.01546v1","created_at":"2026-07-05T11:47:40.888034+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.01546","created_at":"2026-07-05T11:47:40.888034+00:00"},{"alias_kind":"pith_short_12","alias_value":"6CKKRTK7MMMI","created_at":"2026-07-05T11:47:40.888034+00:00"},{"alias_kind":"pith_short_16","alias_value":"6CKKRTK7MMMIBDH5","created_at":"2026-07-05T11:47:40.888034+00:00"},{"alias_kind":"pith_short_8","alias_value":"6CKKRTK7","created_at":"2026-07-05T11:47:40.888034+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.13141","citing_title":"Rethinking RAG in Long Videos: What to Retrieve and How to Use It?","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00983","citing_title":"QCA: Query- and Content-Aware Keyframe Selection for Long Video Understanding","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03301","citing_title":"SagaQA: A Multi-hop Reasoning Benchmark for Long-form Narrative Understanding in TV Series","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05418","citing_title":"VideoStir: Understanding Long Videos via Spatio-Temporally Structured and Intent-Aware RAG","ref_index":43,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6CKKRTK7MMMIBDH5XRBNAREW74","json":"https://pith.science/pith/6CKKRTK7MMMIBDH5XRBNAREW74.json","graph_json":"https://pith.science/api/pith-number/6CKKRTK7MMMIBDH5XRBNAREW74/graph.json","events_json":"https://pith.science/api/pith-number/6CKKRTK7MMMIBDH5XRBNAREW74/events.json","paper":"https://pith.science/paper/6CKKRTK7"},"agent_actions":{"view_html":"https://pith.science/pith/6CKKRTK7MMMIBDH5XRBNAREW74","download_json":"https://pith.science/pith/6CKKRTK7MMMIBDH5XRBNAREW74.json","view_paper":"https://pith.science/paper/6CKKRTK7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.01546&json=true","fetch_graph":"https://pith.science/api/pith-number/6CKKRTK7MMMIBDH5XRBNAREW74/graph.json","fetch_events":"https://pith.science/api/pith-number/6CKKRTK7MMMIBDH5XRBNAREW74/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6CKKRTK7MMMIBDH5XRBNAREW74/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6CKKRTK7MMMIBDH5XRBNAREW74/action/storage_attestation","attest_author":"https://pith.science/pith/6CKKRTK7MMMIBDH5XRBNAREW74/action/author_attestation","sign_citation":"https://pith.science/pith/6CKKRTK7MMMIBDH5XRBNAREW74/action/citation_signature","submit_replication":"https://pith.science/pith/6CKKRTK7MMMIBDH5XRBNAREW74/action/replication_record"}},"created_at":"2026-07-05T11:47:40.888034+00:00","updated_at":"2026-07-05T11:47:40.888034+00:00"}