{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:HEFM6HVP4M55D7WRQPKUR4K4F5","short_pith_number":"pith:HEFM6HVP","schema_version":"1.0","canonical_sha256":"390acf1eafe33bd1fed183d548f15c2f464707b1f52385ec9b4ef44bafb54fcb","source":{"kind":"arxiv","id":"2504.01407","version":3},"attestation_state":"computed","paper":{"title":"ZoomV: Temporal Zoom-in for Efficient Long Video Understanding","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Junwen Pan, Ming Lu, Qi She, Qizhe Zhang, Rui Zhang, Shanghang Zhang, Xin Wan, Yuan Zhang","submitted_at":"2025-04-02T06:47:19Z","abstract_excerpt":"Long video understanding poses a fundamental challenge for large video-language models (LVLMs) due to the overwhelming number of frames and the risk of losing essential context through naive downsampling. Inspired by the way humans watch videos on mobile phones, constantly zooming in on frames of interest, we propose ZoomV, a query-aware temporal zoom-in framework designed for efficient and accurate long video understanding. Specifically, ZoomV operates in three stages: (1) Temporal interests grounding: guided by the query, ZoomV retrieves relevant events and their associated temporal windows "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.01407","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2025-04-02T06:47:19Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"979213f2df739c53303ea516550a9475bc2fcbe24713270bc80e045b7a28b435","abstract_canon_sha256":"862f339230f83b8f15dbf92fa10895fac5339e72bdeb8850b98d06b3a171e103"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-06T01:13:53.120611Z","signature_b64":"vUCgaYE3wRIEK36+pYKWYh7/RZT59TwXNtwl0Wl6BCnQ+CYtLWAuQKnr/Cjqfk0p9dMpURJbX9zw136xrzd1Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"390acf1eafe33bd1fed183d548f15c2f464707b1f52385ec9b4ef44bafb54fcb","last_reissued_at":"2026-08-06T01:13:53.118964Z","signature_status":"signed_v1","first_computed_at":"2026-08-06T01:13:53.118964Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ZoomV: Temporal Zoom-in for Efficient Long Video Understanding","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Junwen Pan, Ming Lu, Qi She, Qizhe Zhang, Rui Zhang, Shanghang Zhang, Xin Wan, Yuan Zhang","submitted_at":"2025-04-02T06:47:19Z","abstract_excerpt":"Long video understanding poses a fundamental challenge for large video-language models (LVLMs) due to the overwhelming number of frames and the risk of losing essential context through naive downsampling. Inspired by the way humans watch videos on mobile phones, constantly zooming in on frames of interest, we propose ZoomV, a query-aware temporal zoom-in framework designed for efficient and accurate long video understanding. Specifically, ZoomV operates in three stages: (1) Temporal interests grounding: guided by the query, ZoomV retrieves relevant events and their associated temporal windows "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.01407","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.01407/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.01407","created_at":"2026-08-06T01:13:53.120414+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.01407v3","created_at":"2026-08-06T01:13:53.120414+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.01407","created_at":"2026-08-06T01:13:53.120414+00:00"},{"alias_kind":"pith_short_12","alias_value":"HEFM6HVP4M55","created_at":"2026-08-06T01:13:53.120414+00:00"},{"alias_kind":"pith_short_16","alias_value":"HEFM6HVP4M55D7WR","created_at":"2026-08-06T01:13:53.120414+00:00"},{"alias_kind":"pith_short_8","alias_value":"HEFM6HVP","created_at":"2026-08-06T01:13:53.120414+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2511.04670","citing_title":"Cambrian-S: Towards Spatial Supersensing in Video","ref_index":101,"is_internal_anchor":true},{"citing_arxiv_id":"2602.17555","citing_title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","ref_index":53,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HEFM6HVP4M55D7WRQPKUR4K4F5","json":"https://pith.science/pith/HEFM6HVP4M55D7WRQPKUR4K4F5.json","graph_json":"https://pith.science/api/pith-number/HEFM6HVP4M55D7WRQPKUR4K4F5/graph.json","events_json":"https://pith.science/api/pith-number/HEFM6HVP4M55D7WRQPKUR4K4F5/events.json","paper":"https://pith.science/paper/HEFM6HVP"},"agent_actions":{"view_html":"https://pith.science/pith/HEFM6HVP4M55D7WRQPKUR4K4F5","download_json":"https://pith.science/pith/HEFM6HVP4M55D7WRQPKUR4K4F5.json","view_paper":"https://pith.science/paper/HEFM6HVP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.01407&json=true","fetch_graph":"https://pith.science/api/pith-number/HEFM6HVP4M55D7WRQPKUR4K4F5/graph.json","fetch_events":"https://pith.science/api/pith-number/HEFM6HVP4M55D7WRQPKUR4K4F5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HEFM6HVP4M55D7WRQPKUR4K4F5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HEFM6HVP4M55D7WRQPKUR4K4F5/action/storage_attestation","attest_author":"https://pith.science/pith/HEFM6HVP4M55D7WRQPKUR4K4F5/action/author_attestation","sign_citation":"https://pith.science/pith/HEFM6HVP4M55D7WRQPKUR4K4F5/action/citation_signature","submit_replication":"https://pith.science/pith/HEFM6HVP4M55D7WRQPKUR4K4F5/action/replication_record"}},"created_at":"2026-08-06T01:13:53.120414+00:00","updated_at":"2026-08-06T01:13:53.120414+00:00"}