{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XGPQR4S772LFRSULS7BK5VWNOL","short_pith_number":"pith:XGPQR4S7","schema_version":"1.0","canonical_sha256":"b99f08f25ffe9658ca8b97c2aed6cd72e15ba834559bfa9285530a3bf3dbc3e5","source":{"kind":"arxiv","id":"2407.03104","version":3},"attestation_state":"computed","paper":{"title":"KeyVideoLLM: Towards Large-scale Video Keyframe Selection","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.MM"],"primary_cat":"cs.CV","authors_text":"Bin Cui, Chong Chen, Conghui He, Hao Liang, Jiapeng Li, Linzhuang Sun, Tianyi Bai, Wentao Zhang, Xijie Huang, Zhengren Wang","submitted_at":"2024-07-03T13:41:44Z","abstract_excerpt":"Recently, with the rise of web videos, managing and understanding large-scale video datasets has become increasingly important. Video Large Language Models (VideoLLMs) have emerged in recent years due to their strong video understanding capabilities. However, training and inference processes for VideoLLMs demand vast amounts of data, presenting significant challenges to data management, particularly regarding efficiency, robustness, and effectiveness. In this work, we present KeyVideoLLM, a text-video frame similarity-based keyframe selection method designed to manage VideoLLM data efficiently"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.03104","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-07-03T13:41:44Z","cross_cats_sorted":["cs.CL","cs.MM"],"title_canon_sha256":"0da793d5b51a4d64de885abe0defb0b44f12871a83fdf3a1642e2b6abe1f9b30","abstract_canon_sha256":"cb912021dd2990ad365666dba0c3ec159aaf500c9b885337b38dd2ccb44f3649"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:54:15.227143Z","signature_b64":"2UVLTQ+d+M4RNCPHOjbMJDTRJb/uXLtDJ6Fe1fh32iU7pOiwfgusF5/FpiCNRPxH+nXA0miD0KiZtKTUif1pBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b99f08f25ffe9658ca8b97c2aed6cd72e15ba834559bfa9285530a3bf3dbc3e5","last_reissued_at":"2026-07-05T08:54:15.226664Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:54:15.226664Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"KeyVideoLLM: Towards Large-scale Video Keyframe Selection","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.MM"],"primary_cat":"cs.CV","authors_text":"Bin Cui, Chong Chen, Conghui He, Hao Liang, Jiapeng Li, Linzhuang Sun, Tianyi Bai, Wentao Zhang, Xijie Huang, Zhengren Wang","submitted_at":"2024-07-03T13:41:44Z","abstract_excerpt":"Recently, with the rise of web videos, managing and understanding large-scale video datasets has become increasingly important. Video Large Language Models (VideoLLMs) have emerged in recent years due to their strong video understanding capabilities. However, training and inference processes for VideoLLMs demand vast amounts of data, presenting significant challenges to data management, particularly regarding efficiency, robustness, and effectiveness. In this work, we present KeyVideoLLM, a text-video frame similarity-based keyframe selection method designed to manage VideoLLM data efficiently"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.03104","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.03104/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.03104","created_at":"2026-07-05T08:54:15.226723+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.03104v3","created_at":"2026-07-05T08:54:15.226723+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.03104","created_at":"2026-07-05T08:54:15.226723+00:00"},{"alias_kind":"pith_short_12","alias_value":"XGPQR4S772LF","created_at":"2026-07-05T08:54:15.226723+00:00"},{"alias_kind":"pith_short_16","alias_value":"XGPQR4S772LFRSUL","created_at":"2026-07-05T08:54:15.226723+00:00"},{"alias_kind":"pith_short_8","alias_value":"XGPQR4S7","created_at":"2026-07-05T08:54:15.226723+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12125","citing_title":"Q-Fold: Query-Aware Focus-Context Spatio-Temporal Folding for Long Video Understanding","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00983","citing_title":"QCA: Query- and Content-Aware Keyframe Selection for Long Video Understanding","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31029","citing_title":"PEEK: Picking Essential frames via Efficient Knowledge distillation","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02569","citing_title":"AdaCodec: A Predictive Visual Code for Video MLLMs","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2506.07180","citing_title":"Flattery in Motion: Benchmarking and Analyzing Sycophancy in Video-LLMs","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11477","citing_title":"LDDR: Linear-DPP-Based Dynamic-Resolution Frame Sampling for Video MLLMs","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05546","citing_title":"Efficient Inference for Large Vision-Language Models: Bottlenecks, Techniques, and Prospects","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04707","citing_title":"OpenWorldLib: A Unified Codebase and Definition of Advanced World Models","ref_index":73,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XGPQR4S772LFRSULS7BK5VWNOL","json":"https://pith.science/pith/XGPQR4S772LFRSULS7BK5VWNOL.json","graph_json":"https://pith.science/api/pith-number/XGPQR4S772LFRSULS7BK5VWNOL/graph.json","events_json":"https://pith.science/api/pith-number/XGPQR4S772LFRSULS7BK5VWNOL/events.json","paper":"https://pith.science/paper/XGPQR4S7"},"agent_actions":{"view_html":"https://pith.science/pith/XGPQR4S772LFRSULS7BK5VWNOL","download_json":"https://pith.science/pith/XGPQR4S772LFRSULS7BK5VWNOL.json","view_paper":"https://pith.science/paper/XGPQR4S7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.03104&json=true","fetch_graph":"https://pith.science/api/pith-number/XGPQR4S772LFRSULS7BK5VWNOL/graph.json","fetch_events":"https://pith.science/api/pith-number/XGPQR4S772LFRSULS7BK5VWNOL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XGPQR4S772LFRSULS7BK5VWNOL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XGPQR4S772LFRSULS7BK5VWNOL/action/storage_attestation","attest_author":"https://pith.science/pith/XGPQR4S772LFRSULS7BK5VWNOL/action/author_attestation","sign_citation":"https://pith.science/pith/XGPQR4S772LFRSULS7BK5VWNOL/action/citation_signature","submit_replication":"https://pith.science/pith/XGPQR4S772LFRSULS7BK5VWNOL/action/replication_record"}},"created_at":"2026-07-05T08:54:15.226723+00:00","updated_at":"2026-07-05T08:54:15.226723+00:00"}