{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:T2BSSTOYVA55KJM5DQYOQMREWK","short_pith_number":"pith:T2BSSTOY","schema_version":"1.0","canonical_sha256":"9e83294dd8a83bd5259d1c30e83224b28600f4b597d142e1ac236fc600f2a0ce","source":{"kind":"arxiv","id":"2410.03226","version":4},"attestation_state":"computed","paper":{"title":"Frame-Voyager: Learning to Query Frames for Video Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Bingni Zhang, Chengkai Jin, Hao Zhang, Huanyu Wang, Jiawei Wu, Qianru Sun, Sheng Jin, Sicheng Yu, Xiaolei Xu, Zhenbang Sun, Zhenghao Chen, Zhongrong Zuo","submitted_at":"2024-10-04T08:26:06Z","abstract_excerpt":"Video Large Language Models (Video-LLMs) have made remarkable progress in video understanding tasks. However, they are constrained by the maximum length of input tokens, making it impractical to input entire videos. Existing frame selection approaches, such as uniform frame sampling and text-frame retrieval, fail to account for the information density variations in the videos or the complex instructions in the tasks, leading to sub-optimal performance. In this paper, we propose Frame-Voyager that learns to query informative frame combinations, based on the given textual queries in the task. To"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.03226","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-10-04T08:26:06Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"a22cd97cb5b3d4bb5cca179234b323eb5513fbaf385cfefe76896a66261284f6","abstract_canon_sha256":"4bc06031f3a493b34287b6ba7cb8dcef395487d2d94360031ff20050357c3890"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:40:36.062467Z","signature_b64":"3uv5ABRwlWFiHdUESer741fc04x91gttEbttXMsKxxL3omMwr+tfLjQhDwx3GJAxUcd6gqY+E4f2tPV/l+eJAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9e83294dd8a83bd5259d1c30e83224b28600f4b597d142e1ac236fc600f2a0ce","last_reissued_at":"2026-07-05T10:40:36.061944Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:40:36.061944Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Frame-Voyager: Learning to Query Frames for Video Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Bingni Zhang, Chengkai Jin, Hao Zhang, Huanyu Wang, Jiawei Wu, Qianru Sun, Sheng Jin, Sicheng Yu, Xiaolei Xu, Zhenbang Sun, Zhenghao Chen, Zhongrong Zuo","submitted_at":"2024-10-04T08:26:06Z","abstract_excerpt":"Video Large Language Models (Video-LLMs) have made remarkable progress in video understanding tasks. However, they are constrained by the maximum length of input tokens, making it impractical to input entire videos. Existing frame selection approaches, such as uniform frame sampling and text-frame retrieval, fail to account for the information density variations in the videos or the complex instructions in the tasks, leading to sub-optimal performance. In this paper, we propose Frame-Voyager that learns to query informative frame combinations, based on the given textual queries in the task. To"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.03226","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.03226/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.03226","created_at":"2026-07-05T10:40:36.062005+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.03226v4","created_at":"2026-07-05T10:40:36.062005+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.03226","created_at":"2026-07-05T10:40:36.062005+00:00"},{"alias_kind":"pith_short_12","alias_value":"T2BSSTOYVA55","created_at":"2026-07-05T10:40:36.062005+00:00"},{"alias_kind":"pith_short_16","alias_value":"T2BSSTOYVA55KJM5","created_at":"2026-07-05T10:40:36.062005+00:00"},{"alias_kind":"pith_short_8","alias_value":"T2BSSTOY","created_at":"2026-07-05T10:40:36.062005+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01737","citing_title":"ReQuest: Rethinking-based Question-Aware Frame Selection for Long-Form Video QA","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31029","citing_title":"PEEK: Picking Essential frames via Efficient Knowledge distillation","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09223","citing_title":"CREST: Curvature-Regulated Event-Centric Sampling for Efficient Long-Video Understanding","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22678","citing_title":"Swift Sampling: Selecting Temporal Surprises via Taylor Series","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10762","citing_title":"GridProbe: Posterior-Probing for Adaptive Test-Time Compute in Long-Video VLMs","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09874","citing_title":"EgoMemReason: A Memory-Driven Reasoning Benchmark for Long-Horizon Egocentric Video Understanding","ref_index":81,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09223","citing_title":"CREST: Curvature-Regulated Event-Centric Sampling for Efficient Long-Video Understanding","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00444","citing_title":"Scaling Video Understanding via Compact Latent Multi-Agent Collaboration","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05546","citing_title":"Efficient Inference for Large Vision-Language Models: Bottlenecks, Techniques, and Prospects","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17052","citing_title":"OASIS: On-Demand Hierarchical Event Memory for Streaming Video Reasoning","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17422","citing_title":"Where to Focus: Query-Modulated Multimodal Keyframe Selection for Long Video Understanding","ref_index":43,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/T2BSSTOYVA55KJM5DQYOQMREWK","json":"https://pith.science/pith/T2BSSTOYVA55KJM5DQYOQMREWK.json","graph_json":"https://pith.science/api/pith-number/T2BSSTOYVA55KJM5DQYOQMREWK/graph.json","events_json":"https://pith.science/api/pith-number/T2BSSTOYVA55KJM5DQYOQMREWK/events.json","paper":"https://pith.science/paper/T2BSSTOY"},"agent_actions":{"view_html":"https://pith.science/pith/T2BSSTOYVA55KJM5DQYOQMREWK","download_json":"https://pith.science/pith/T2BSSTOYVA55KJM5DQYOQMREWK.json","view_paper":"https://pith.science/paper/T2BSSTOY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.03226&json=true","fetch_graph":"https://pith.science/api/pith-number/T2BSSTOYVA55KJM5DQYOQMREWK/graph.json","fetch_events":"https://pith.science/api/pith-number/T2BSSTOYVA55KJM5DQYOQMREWK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/T2BSSTOYVA55KJM5DQYOQMREWK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/T2BSSTOYVA55KJM5DQYOQMREWK/action/storage_attestation","attest_author":"https://pith.science/pith/T2BSSTOYVA55KJM5DQYOQMREWK/action/author_attestation","sign_citation":"https://pith.science/pith/T2BSSTOYVA55KJM5DQYOQMREWK/action/citation_signature","submit_replication":"https://pith.science/pith/T2BSSTOYVA55KJM5DQYOQMREWK/action/replication_record"}},"created_at":"2026-07-05T10:40:36.062005+00:00","updated_at":"2026-07-05T10:40:36.062005+00:00"}