{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:NVZNTJBJILAVX3K7CC45OYBF2P","short_pith_number":"pith:NVZNTJBJ","schema_version":"1.0","canonical_sha256":"6d72d9a42942c15bed5f10b9d76025d3f58bf4c2fc38ba2d669d48670bfefda6","source":{"kind":"arxiv","id":"2508.08989","version":1},"attestation_state":"computed","paper":{"title":"KFFocus: Highlighting Keyframes for Enhanced Video Understanding","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chunwei Wang, Hang Xu, Li Zhang, Ming Nie","submitted_at":"2025-08-12T14:57:03Z","abstract_excerpt":"Recently, with the emergence of large language models, multimodal LLMs have demonstrated exceptional capabilities in image and video modalities. Despite advancements in video comprehension, the substantial computational demands of long video sequences lead current video LLMs (Vid-LLMs) to employ compression strategies at both the inter-frame level (e.g., uniform sampling of video frames) and intra-frame level (e.g., condensing all visual tokens of each frame into a limited number). However, this approach often neglects the uneven temporal distribution of critical information across frames, ris"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.08989","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2025-08-12T14:57:03Z","cross_cats_sorted":[],"title_canon_sha256":"e6ee3fd33784d676e35672f2a4e1d2166ca2b879573e57ce9d24614f097fdbb7","abstract_canon_sha256":"0cb2583651aa173efd47eee5b6c7f66e7d2ef4184a9ceaab8d34b33d80469758"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:52:37.958699Z","signature_b64":"nUm5GHwMuXd/0p9Z1G0quFYLTCYfHzstm1EfkSAP9EmseFbvFxtdj4V+pgOeULtNw5Ic7z/BPYNjHULmIAG5CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6d72d9a42942c15bed5f10b9d76025d3f58bf4c2fc38ba2d669d48670bfefda6","last_reissued_at":"2026-07-05T11:52:37.958203Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:52:37.958203Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"KFFocus: Highlighting Keyframes for Enhanced Video Understanding","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chunwei Wang, Hang Xu, Li Zhang, Ming Nie","submitted_at":"2025-08-12T14:57:03Z","abstract_excerpt":"Recently, with the emergence of large language models, multimodal LLMs have demonstrated exceptional capabilities in image and video modalities. Despite advancements in video comprehension, the substantial computational demands of long video sequences lead current video LLMs (Vid-LLMs) to employ compression strategies at both the inter-frame level (e.g., uniform sampling of video frames) and intra-frame level (e.g., condensing all visual tokens of each frame into a limited number). However, this approach often neglects the uneven temporal distribution of critical information across frames, ris"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.08989","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.08989/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.08989","created_at":"2026-07-05T11:52:37.958261+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.08989v1","created_at":"2026-07-05T11:52:37.958261+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.08989","created_at":"2026-07-05T11:52:37.958261+00:00"},{"alias_kind":"pith_short_12","alias_value":"NVZNTJBJILAV","created_at":"2026-07-05T11:52:37.958261+00:00"},{"alias_kind":"pith_short_16","alias_value":"NVZNTJBJILAVX3K7","created_at":"2026-07-05T11:52:37.958261+00:00"},{"alias_kind":"pith_short_8","alias_value":"NVZNTJBJ","created_at":"2026-07-05T11:52:37.958261+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19849","citing_title":"ViCoStream: Streaming VideoLLMs Can Run Beyond 100 FPS with Stage-Wise Coordinated Inference","ref_index":43,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NVZNTJBJILAVX3K7CC45OYBF2P","json":"https://pith.science/pith/NVZNTJBJILAVX3K7CC45OYBF2P.json","graph_json":"https://pith.science/api/pith-number/NVZNTJBJILAVX3K7CC45OYBF2P/graph.json","events_json":"https://pith.science/api/pith-number/NVZNTJBJILAVX3K7CC45OYBF2P/events.json","paper":"https://pith.science/paper/NVZNTJBJ"},"agent_actions":{"view_html":"https://pith.science/pith/NVZNTJBJILAVX3K7CC45OYBF2P","download_json":"https://pith.science/pith/NVZNTJBJILAVX3K7CC45OYBF2P.json","view_paper":"https://pith.science/paper/NVZNTJBJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.08989&json=true","fetch_graph":"https://pith.science/api/pith-number/NVZNTJBJILAVX3K7CC45OYBF2P/graph.json","fetch_events":"https://pith.science/api/pith-number/NVZNTJBJILAVX3K7CC45OYBF2P/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NVZNTJBJILAVX3K7CC45OYBF2P/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NVZNTJBJILAVX3K7CC45OYBF2P/action/storage_attestation","attest_author":"https://pith.science/pith/NVZNTJBJILAVX3K7CC45OYBF2P/action/author_attestation","sign_citation":"https://pith.science/pith/NVZNTJBJILAVX3K7CC45OYBF2P/action/citation_signature","submit_replication":"https://pith.science/pith/NVZNTJBJILAVX3K7CC45OYBF2P/action/replication_record"}},"created_at":"2026-07-05T11:52:37.958261+00:00","updated_at":"2026-07-05T11:52:37.958261+00:00"}