{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZVYJABDV3DIKI4WMDVSBBY2J5D","short_pith_number":"pith:ZVYJABDV","schema_version":"1.0","canonical_sha256":"cd70900475d8d0a472cc1d6410e349e8e6fbe1f7895dec4f4a706039f812b0ca","source":{"kind":"arxiv","id":"2501.05874","version":3},"attestation_state":"computed","paper":{"title":"VideoRAG: Retrieval-Augmented Generation over Video Corpus","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.IR","cs.LG"],"primary_cat":"cs.CV","authors_text":"Jinheon Baek, Kangsan Kim, Soyeong Jeong, Sung Ju Hwang","submitted_at":"2025-01-10T11:17:15Z","abstract_excerpt":"Retrieval-Augmented Generation (RAG) is a powerful strategy for improving the factual accuracy of models by retrieving external knowledge relevant to queries and incorporating it into the generation process. However, existing approaches primarily focus on text, with some recent advancements considering images, and they largely overlook videos, a rich source of multimodal knowledge capable of representing contextual details more effectively than any other modality. While very recent studies explore the use of videos in response generation, they either predefine query-associated videos without r"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.05874","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-01-10T11:17:15Z","cross_cats_sorted":["cs.AI","cs.CL","cs.IR","cs.LG"],"title_canon_sha256":"8c1c1633d0eb19198542e2aa0d3fbc9c903715e68421fc9807742fa247863213","abstract_canon_sha256":"f3900e06f6fa070e976ebbee0ec29ad7d7832cd8767c59aba83d455f24472f30"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:11:39.602130Z","signature_b64":"7zMFJIu58Fr4rUN+r+P0fEGut84vAg7Ot7qNyMF23w/gBVZYJc4oVUdIVu4nVgU7Gr2al4rWmPxDHTT7qJ7fDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cd70900475d8d0a472cc1d6410e349e8e6fbe1f7895dec4f4a706039f812b0ca","last_reissued_at":"2026-07-05T11:11:39.601597Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:11:39.601597Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VideoRAG: Retrieval-Augmented Generation over Video Corpus","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.IR","cs.LG"],"primary_cat":"cs.CV","authors_text":"Jinheon Baek, Kangsan Kim, Soyeong Jeong, Sung Ju Hwang","submitted_at":"2025-01-10T11:17:15Z","abstract_excerpt":"Retrieval-Augmented Generation (RAG) is a powerful strategy for improving the factual accuracy of models by retrieving external knowledge relevant to queries and incorporating it into the generation process. However, existing approaches primarily focus on text, with some recent advancements considering images, and they largely overlook videos, a rich source of multimodal knowledge capable of representing contextual details more effectively than any other modality. While very recent studies explore the use of videos in response generation, they either predefine query-associated videos without r"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.05874","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.05874/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.05874","created_at":"2026-07-05T11:11:39.601653+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.05874v3","created_at":"2026-07-05T11:11:39.601653+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.05874","created_at":"2026-07-05T11:11:39.601653+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZVYJABDV3DIK","created_at":"2026-07-05T11:11:39.601653+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZVYJABDV3DIKI4WM","created_at":"2026-07-05T11:11:39.601653+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZVYJABDV","created_at":"2026-07-05T11:11:39.601653+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21734","citing_title":"HPP: Hierarchical Programmatic Probing for Long Video Understanding by Decoupling Perception and Reasoning","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06532","citing_title":"GOPAgen: Motion-Aware and Efficient Agentic Long-Video Understanding with Structural Memory and Hierarchical Reasoning","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05418","citing_title":"VideoStir: Understanding Long Videos via Spatio-Temporally Structured and Intent-Aware RAG","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04372","citing_title":"Graph-to-Frame RAG: Visual-Space Knowledge Fusion for Training-Free and Auditable Video Reasoning","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15225","citing_title":"UrbanClipAtlas: A Visual Analytics Framework for Event and Scene Retrieval in Urban Videos","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZVYJABDV3DIKI4WMDVSBBY2J5D","json":"https://pith.science/pith/ZVYJABDV3DIKI4WMDVSBBY2J5D.json","graph_json":"https://pith.science/api/pith-number/ZVYJABDV3DIKI4WMDVSBBY2J5D/graph.json","events_json":"https://pith.science/api/pith-number/ZVYJABDV3DIKI4WMDVSBBY2J5D/events.json","paper":"https://pith.science/paper/ZVYJABDV"},"agent_actions":{"view_html":"https://pith.science/pith/ZVYJABDV3DIKI4WMDVSBBY2J5D","download_json":"https://pith.science/pith/ZVYJABDV3DIKI4WMDVSBBY2J5D.json","view_paper":"https://pith.science/paper/ZVYJABDV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.05874&json=true","fetch_graph":"https://pith.science/api/pith-number/ZVYJABDV3DIKI4WMDVSBBY2J5D/graph.json","fetch_events":"https://pith.science/api/pith-number/ZVYJABDV3DIKI4WMDVSBBY2J5D/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZVYJABDV3DIKI4WMDVSBBY2J5D/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZVYJABDV3DIKI4WMDVSBBY2J5D/action/storage_attestation","attest_author":"https://pith.science/pith/ZVYJABDV3DIKI4WMDVSBBY2J5D/action/author_attestation","sign_citation":"https://pith.science/pith/ZVYJABDV3DIKI4WMDVSBBY2J5D/action/citation_signature","submit_replication":"https://pith.science/pith/ZVYJABDV3DIKI4WMDVSBBY2J5D/action/replication_record"}},"created_at":"2026-07-05T11:11:39.601653+00:00","updated_at":"2026-07-05T11:11:39.601653+00:00"}