{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PBED2RE5AWK2EUCUJNQTSGGJHJ","short_pith_number":"pith:PBED2RE5","schema_version":"1.0","canonical_sha256":"78483d449d0595a250544b613918c93a47bed864a917c241a1308d049f2112c5","source":{"kind":"arxiv","id":"2406.12846","version":2},"attestation_state":"computed","paper":{"title":"DrVideo: Document Retrieval Based Long Video Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bin Sun, Chenhui Gou, Hamid Rezatofighi, Hengcan Shi, Jianfei Cai, Shutao Li, Ziyu Ma","submitted_at":"2024-06-18T17:59:03Z","abstract_excerpt":"Most of the existing methods for video understanding primarily focus on videos only lasting tens of seconds, with limited exploration of techniques for handling long videos. The increased number of frames in long videos poses two main challenges: difficulty in locating key information and performing long-range reasoning. Thus, we propose DrVideo, a document-retrieval-based system designed for long video understanding. Our key idea is to convert the long-video understanding problem into a long-document understanding task so as to effectively leverage the power of large language models. Specific"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.12846","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-06-18T17:59:03Z","cross_cats_sorted":[],"title_canon_sha256":"de4af032233b4228f94be9df8a1c589916ea5638770c713aa14a2d8238d1fa08","abstract_canon_sha256":"fdad0067b872ea460819218b214ac501ce0fe27b9a833d2a64e838f0ff1fd9f8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:39:24.208934Z","signature_b64":"sswStK2ojnFK6SIxS8iGZTxqRMFfuYvsucY+998XPpgbVZIVUn3HEmov48uEu3jEpN+kc/lCn6gz5UWe9LiQCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"78483d449d0595a250544b613918c93a47bed864a917c241a1308d049f2112c5","last_reissued_at":"2026-07-05T09:39:24.208467Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:39:24.208467Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DrVideo: Document Retrieval Based Long Video Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bin Sun, Chenhui Gou, Hamid Rezatofighi, Hengcan Shi, Jianfei Cai, Shutao Li, Ziyu Ma","submitted_at":"2024-06-18T17:59:03Z","abstract_excerpt":"Most of the existing methods for video understanding primarily focus on videos only lasting tens of seconds, with limited exploration of techniques for handling long videos. The increased number of frames in long videos poses two main challenges: difficulty in locating key information and performing long-range reasoning. Thus, we propose DrVideo, a document-retrieval-based system designed for long video understanding. Our key idea is to convert the long-video understanding problem into a long-document understanding task so as to effectively leverage the power of large language models. Specific"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.12846","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.12846/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.12846","created_at":"2026-07-05T09:39:24.208524+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.12846v2","created_at":"2026-07-05T09:39:24.208524+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.12846","created_at":"2026-07-05T09:39:24.208524+00:00"},{"alias_kind":"pith_short_12","alias_value":"PBED2RE5AWK2","created_at":"2026-07-05T09:39:24.208524+00:00"},{"alias_kind":"pith_short_16","alias_value":"PBED2RE5AWK2EUCU","created_at":"2026-07-05T09:39:24.208524+00:00"},{"alias_kind":"pith_short_8","alias_value":"PBED2RE5","created_at":"2026-07-05T09:39:24.208524+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00816","citing_title":"Towards High-Resolution Visual Perception via Hierarchical Entity Exploration","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2509.24943","citing_title":"Perceive, Verify and Understand Long Video: Multi-Granular Perception and Active Verification via Interactive Agents","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2512.08410","citing_title":"Towards Effective Long Video Understanding of Multimodal Large Language Models via One-shot Clip Retrieval","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02891","citing_title":"Progressive Video Condensation with MLLM Agent for Long-form Video Understanding","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PBED2RE5AWK2EUCUJNQTSGGJHJ","json":"https://pith.science/pith/PBED2RE5AWK2EUCUJNQTSGGJHJ.json","graph_json":"https://pith.science/api/pith-number/PBED2RE5AWK2EUCUJNQTSGGJHJ/graph.json","events_json":"https://pith.science/api/pith-number/PBED2RE5AWK2EUCUJNQTSGGJHJ/events.json","paper":"https://pith.science/paper/PBED2RE5"},"agent_actions":{"view_html":"https://pith.science/pith/PBED2RE5AWK2EUCUJNQTSGGJHJ","download_json":"https://pith.science/pith/PBED2RE5AWK2EUCUJNQTSGGJHJ.json","view_paper":"https://pith.science/paper/PBED2RE5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.12846&json=true","fetch_graph":"https://pith.science/api/pith-number/PBED2RE5AWK2EUCUJNQTSGGJHJ/graph.json","fetch_events":"https://pith.science/api/pith-number/PBED2RE5AWK2EUCUJNQTSGGJHJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PBED2RE5AWK2EUCUJNQTSGGJHJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PBED2RE5AWK2EUCUJNQTSGGJHJ/action/storage_attestation","attest_author":"https://pith.science/pith/PBED2RE5AWK2EUCUJNQTSGGJHJ/action/author_attestation","sign_citation":"https://pith.science/pith/PBED2RE5AWK2EUCUJNQTSGGJHJ/action/citation_signature","submit_replication":"https://pith.science/pith/PBED2RE5AWK2EUCUJNQTSGGJHJ/action/replication_record"}},"created_at":"2026-07-05T09:39:24.208524+00:00","updated_at":"2026-07-05T09:39:24.208524+00:00"}