{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:WUMMBWJRE6ZRFRT7L6RQSGE7KG","short_pith_number":"pith:WUMMBWJR","schema_version":"1.0","canonical_sha256":"b518c0d93127b312c67f5fa309189f519b7e99d121182957cf8d5acd88d9de8c","source":{"kind":"arxiv","id":"2506.22139","version":3},"attestation_state":"computed","paper":{"title":"Q-Frame: Query-aware Frame Selection and Multi-Resolution Adaptation for Video-LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiahui Yang, Jian Luan, Jianqin Yin, Shaojie Zhang, Zhenbo Luo","submitted_at":"2025-06-27T11:30:51Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have demonstrated significant success in visual understanding tasks. However, challenges persist in adapting these models for video comprehension due to the large volume of data and temporal complexity. Existing Video-LLMs using uniform frame sampling often struggle to capture the query-related crucial spatiotemporal clues of videos effectively. In this paper, we introduce Q-Frame, a novel approach for adaptive frame selection and multi-resolution scaling tailored to the video's content and the specific query. Q-Frame employs a training-free, plug-and-p"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.22139","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-06-27T11:30:51Z","cross_cats_sorted":[],"title_canon_sha256":"b1ac7835809bad633fe33c7cb9c0d09f357c4ee2c63d4599740babca4fbd0c8f","abstract_canon_sha256":"d97803d793f4fb739dd8cfb49ce34838650cb302307e28377e1aa6ff5c986f43"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:40:55.554090Z","signature_b64":"o4JFEZ4RNiMaEEW4NYjRraM9sVDmMr6jj44mC4re+ev9j+9YS7uMb1wAO8TBC07+6uPRutn9pjPHvhKhcXLrDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b518c0d93127b312c67f5fa309189f519b7e99d121182957cf8d5acd88d9de8c","last_reissued_at":"2026-07-05T11:40:55.553569Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:40:55.553569Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Q-Frame: Query-aware Frame Selection and Multi-Resolution Adaptation for Video-LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiahui Yang, Jian Luan, Jianqin Yin, Shaojie Zhang, Zhenbo Luo","submitted_at":"2025-06-27T11:30:51Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have demonstrated significant success in visual understanding tasks. However, challenges persist in adapting these models for video comprehension due to the large volume of data and temporal complexity. Existing Video-LLMs using uniform frame sampling often struggle to capture the query-related crucial spatiotemporal clues of videos effectively. In this paper, we introduce Q-Frame, a novel approach for adaptive frame selection and multi-resolution scaling tailored to the video's content and the specific query. Q-Frame employs a training-free, plug-and-p"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.22139","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.22139/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.22139","created_at":"2026-07-05T11:40:55.553629+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.22139v3","created_at":"2026-07-05T11:40:55.553629+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.22139","created_at":"2026-07-05T11:40:55.553629+00:00"},{"alias_kind":"pith_short_12","alias_value":"WUMMBWJRE6ZR","created_at":"2026-07-05T11:40:55.553629+00:00"},{"alias_kind":"pith_short_16","alias_value":"WUMMBWJRE6ZRFRT7","created_at":"2026-07-05T11:40:55.553629+00:00"},{"alias_kind":"pith_short_8","alias_value":"WUMMBWJR","created_at":"2026-07-05T11:40:55.553629+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24187","citing_title":"Towards Fast and Effective Long Video Understanding of Multimodal Large Language Models via Adaptive Quasi-Gaussian Sampling","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07433","citing_title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00983","citing_title":"QCA: Query- and Content-Aware Keyframe Selection for Long Video Understanding","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03100","citing_title":"Zero-Shot 3D Question Answering via Hierarchical View-to-Token Transportation","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31029","citing_title":"PEEK: Picking Essential frames via Efficient Knowledge distillation","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08077","citing_title":"AdaSpark: Adaptive Sparsity for Efficient Long-Video Understanding","ref_index":50,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WUMMBWJRE6ZRFRT7L6RQSGE7KG","json":"https://pith.science/pith/WUMMBWJRE6ZRFRT7L6RQSGE7KG.json","graph_json":"https://pith.science/api/pith-number/WUMMBWJRE6ZRFRT7L6RQSGE7KG/graph.json","events_json":"https://pith.science/api/pith-number/WUMMBWJRE6ZRFRT7L6RQSGE7KG/events.json","paper":"https://pith.science/paper/WUMMBWJR"},"agent_actions":{"view_html":"https://pith.science/pith/WUMMBWJRE6ZRFRT7L6RQSGE7KG","download_json":"https://pith.science/pith/WUMMBWJRE6ZRFRT7L6RQSGE7KG.json","view_paper":"https://pith.science/paper/WUMMBWJR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.22139&json=true","fetch_graph":"https://pith.science/api/pith-number/WUMMBWJRE6ZRFRT7L6RQSGE7KG/graph.json","fetch_events":"https://pith.science/api/pith-number/WUMMBWJRE6ZRFRT7L6RQSGE7KG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WUMMBWJRE6ZRFRT7L6RQSGE7KG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WUMMBWJRE6ZRFRT7L6RQSGE7KG/action/storage_attestation","attest_author":"https://pith.science/pith/WUMMBWJRE6ZRFRT7L6RQSGE7KG/action/author_attestation","sign_citation":"https://pith.science/pith/WUMMBWJRE6ZRFRT7L6RQSGE7KG/action/citation_signature","submit_replication":"https://pith.science/pith/WUMMBWJRE6ZRFRT7L6RQSGE7KG/action/replication_record"}},"created_at":"2026-07-05T11:40:55.553629+00:00","updated_at":"2026-07-05T11:40:55.553629+00:00"}