{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:7WJYKAXCF4PDZOKC44FIULNNYN","short_pith_number":"pith:7WJYKAXC","schema_version":"1.0","canonical_sha256":"fd938502e22f1e3cb942e70a8a2dadc376fdd069cf736ec31bf2ebe0e5ec95b5","source":{"kind":"arxiv","id":"2505.03173","version":1},"attestation_state":"computed","paper":{"title":"RAVU: Retrieval Augmented Video Understanding with Compositional Reasoning over Graph","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Ayush Singh, Dishank Aggarwal, Moyuru Yamada, Sameer Malik","submitted_at":"2025-05-06T04:38:09Z","abstract_excerpt":"Comprehending long videos remains a significant challenge for Large Multi-modal Models (LMMs). Current LMMs struggle to process even minutes to hours videos due to their lack of explicit memory and retrieval mechanisms. To address this limitation, we propose RAVU (Retrieval Augmented Video Understanding), a novel framework for video understanding enhanced by retrieval with compositional reasoning over a spatio-temporal graph. We construct a graph representation of the video, capturing both spatial and temporal relationships between entities. This graph serves as a long-term memory, allowing us"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.03173","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-05-06T04:38:09Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"422eb89e7438d523079f0c03037da156bc6af5597d58291b5a86c0bffc3370f3","abstract_canon_sha256":"d499e39801a014f42c05d634f78e713c16ecb5c83ae8289527937e24548691d6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:59:11.453158Z","signature_b64":"SaG3kh9K748VN0cTF+n7e3MRcOsWHpTZdM6UMwzU6/hLXs8In50nChENVPu8Dd6W4hUFXNqfwDaOfuEdQ8euCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fd938502e22f1e3cb942e70a8a2dadc376fdd069cf736ec31bf2ebe0e5ec95b5","last_reissued_at":"2026-07-05T10:59:11.452693Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:59:11.452693Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RAVU: Retrieval Augmented Video Understanding with Compositional Reasoning over Graph","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Ayush Singh, Dishank Aggarwal, Moyuru Yamada, Sameer Malik","submitted_at":"2025-05-06T04:38:09Z","abstract_excerpt":"Comprehending long videos remains a significant challenge for Large Multi-modal Models (LMMs). Current LMMs struggle to process even minutes to hours videos due to their lack of explicit memory and retrieval mechanisms. To address this limitation, we propose RAVU (Retrieval Augmented Video Understanding), a novel framework for video understanding enhanced by retrieval with compositional reasoning over a spatio-temporal graph. We construct a graph representation of the video, capturing both spatial and temporal relationships between entities. This graph serves as a long-term memory, allowing us"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.03173","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.03173/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.03173","created_at":"2026-07-05T10:59:11.452749+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.03173v1","created_at":"2026-07-05T10:59:11.452749+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.03173","created_at":"2026-07-05T10:59:11.452749+00:00"},{"alias_kind":"pith_short_12","alias_value":"7WJYKAXCF4PD","created_at":"2026-07-05T10:59:11.452749+00:00"},{"alias_kind":"pith_short_16","alias_value":"7WJYKAXCF4PDZOKC","created_at":"2026-07-05T10:59:11.452749+00:00"},{"alias_kind":"pith_short_8","alias_value":"7WJYKAXC","created_at":"2026-07-05T10:59:11.452749+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.13141","citing_title":"Rethinking RAG in Long Videos: What to Retrieve and How to Use It?","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7WJYKAXCF4PDZOKC44FIULNNYN","json":"https://pith.science/pith/7WJYKAXCF4PDZOKC44FIULNNYN.json","graph_json":"https://pith.science/api/pith-number/7WJYKAXCF4PDZOKC44FIULNNYN/graph.json","events_json":"https://pith.science/api/pith-number/7WJYKAXCF4PDZOKC44FIULNNYN/events.json","paper":"https://pith.science/paper/7WJYKAXC"},"agent_actions":{"view_html":"https://pith.science/pith/7WJYKAXCF4PDZOKC44FIULNNYN","download_json":"https://pith.science/pith/7WJYKAXCF4PDZOKC44FIULNNYN.json","view_paper":"https://pith.science/paper/7WJYKAXC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.03173&json=true","fetch_graph":"https://pith.science/api/pith-number/7WJYKAXCF4PDZOKC44FIULNNYN/graph.json","fetch_events":"https://pith.science/api/pith-number/7WJYKAXCF4PDZOKC44FIULNNYN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7WJYKAXCF4PDZOKC44FIULNNYN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7WJYKAXCF4PDZOKC44FIULNNYN/action/storage_attestation","attest_author":"https://pith.science/pith/7WJYKAXCF4PDZOKC44FIULNNYN/action/author_attestation","sign_citation":"https://pith.science/pith/7WJYKAXCF4PDZOKC44FIULNNYN/action/citation_signature","submit_replication":"https://pith.science/pith/7WJYKAXCF4PDZOKC44FIULNNYN/action/replication_record"}},"created_at":"2026-07-05T10:59:11.452749+00:00","updated_at":"2026-07-05T10:59:11.452749+00:00"}