{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:V2OZ3MRACWS3USJXC7OB73A6LV","short_pith_number":"pith:V2OZ3MRA","schema_version":"1.0","canonical_sha256":"ae9d9db22015a5ba493717dc1fec1e5d6e5504068fb50b245cee3476778970d2","source":{"kind":"arxiv","id":"2405.19209","version":3},"attestation_state":"computed","paper":{"title":"VideoTree: Adaptive Tree-based Video Representation for LLM Reasoning on Long Videos","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Elias Stengel-Eskin, Feng Cheng, Gedas Bertasius, Jaehong Yoon, Mohit Bansal, Shoubin Yu, Ziyang Wang","submitted_at":"2024-05-29T15:49:09Z","abstract_excerpt":"Long-form video understanding is complicated by the high redundancy of video data and the abundance of query-irrelevant information. To tackle these challenges, we propose VideoTree, a training-free framework which builds a query-adaptive and hierarchical video representation for LLM reasoning over long-form videos. First, VideoTree extracts query-relevant information from the input video through an iterative process, progressively refining the selection of keyframes based on their relevance to the query. Furthermore, VideoTree leverages the inherent hierarchical structure of long video data, "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.19209","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-05-29T15:49:09Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"ea726fef9d12fd76eea53c3860e9c7a084be8069d7ac4b41889df9642965225a","abstract_canon_sha256":"f70960af7b433189f4047a89e826664e0caa6605a82ddbb0464b6df4dbfaa3a9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:31:23.269183Z","signature_b64":"GH2hkUbOiEfKX92+3riwjVm4n8jK2jSXDecOHS1U+3Unxwb7rErv4kuAHiPd7UlTD9UTbyoXDHOE6SPs1q4zCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ae9d9db22015a5ba493717dc1fec1e5d6e5504068fb50b245cee3476778970d2","last_reissued_at":"2026-07-05T10:31:23.268549Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:31:23.268549Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VideoTree: Adaptive Tree-based Video Representation for LLM Reasoning on Long Videos","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Elias Stengel-Eskin, Feng Cheng, Gedas Bertasius, Jaehong Yoon, Mohit Bansal, Shoubin Yu, Ziyang Wang","submitted_at":"2024-05-29T15:49:09Z","abstract_excerpt":"Long-form video understanding is complicated by the high redundancy of video data and the abundance of query-irrelevant information. To tackle these challenges, we propose VideoTree, a training-free framework which builds a query-adaptive and hierarchical video representation for LLM reasoning over long-form videos. First, VideoTree extracts query-relevant information from the input video through an iterative process, progressively refining the selection of keyframes based on their relevance to the query. Furthermore, VideoTree leverages the inherent hierarchical structure of long video data, "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.19209","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.19209/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.19209","created_at":"2026-07-05T10:31:23.268631+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.19209v3","created_at":"2026-07-05T10:31:23.268631+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.19209","created_at":"2026-07-05T10:31:23.268631+00:00"},{"alias_kind":"pith_short_12","alias_value":"V2OZ3MRACWS3","created_at":"2026-07-05T10:31:23.268631+00:00"},{"alias_kind":"pith_short_16","alias_value":"V2OZ3MRACWS3USJX","created_at":"2026-07-05T10:31:23.268631+00:00"},{"alias_kind":"pith_short_8","alias_value":"V2OZ3MRA","created_at":"2026-07-05T10:31:23.268631+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2504.09583","citing_title":"AirVista-II: An Agentic System for Embodied UAVs Toward Dynamic Scene Semantic Understanding","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2601.14724","citing_title":"HERMES: KV Cache as Hierarchical Memory for Efficient Streaming Video Understanding","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02891","citing_title":"Progressive Video Condensation with MLLM Agent for Long-form Video Understanding","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05079","citing_title":"SVAgent: Storyline-Guided Long Video Understanding via Cross-Modal Multi-Agent Collaboration","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14149","citing_title":"One Token per Highly Selective Frame: Towards Extreme Compression for Long Video Understanding","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01662","citing_title":"Video Active Perception: Effective Inference-Time Long-Form Video Understanding with Vision-Language Models","ref_index":47,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/V2OZ3MRACWS3USJXC7OB73A6LV","json":"https://pith.science/pith/V2OZ3MRACWS3USJXC7OB73A6LV.json","graph_json":"https://pith.science/api/pith-number/V2OZ3MRACWS3USJXC7OB73A6LV/graph.json","events_json":"https://pith.science/api/pith-number/V2OZ3MRACWS3USJXC7OB73A6LV/events.json","paper":"https://pith.science/paper/V2OZ3MRA"},"agent_actions":{"view_html":"https://pith.science/pith/V2OZ3MRACWS3USJXC7OB73A6LV","download_json":"https://pith.science/pith/V2OZ3MRACWS3USJXC7OB73A6LV.json","view_paper":"https://pith.science/paper/V2OZ3MRA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.19209&json=true","fetch_graph":"https://pith.science/api/pith-number/V2OZ3MRACWS3USJXC7OB73A6LV/graph.json","fetch_events":"https://pith.science/api/pith-number/V2OZ3MRACWS3USJXC7OB73A6LV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/V2OZ3MRACWS3USJXC7OB73A6LV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/V2OZ3MRACWS3USJXC7OB73A6LV/action/storage_attestation","attest_author":"https://pith.science/pith/V2OZ3MRACWS3USJXC7OB73A6LV/action/author_attestation","sign_citation":"https://pith.science/pith/V2OZ3MRACWS3USJXC7OB73A6LV/action/citation_signature","submit_replication":"https://pith.science/pith/V2OZ3MRACWS3USJXC7OB73A6LV/action/replication_record"}},"created_at":"2026-07-05T10:31:23.268631+00:00","updated_at":"2026-07-05T10:31:23.268631+00:00"}