{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:2SH2B4O5BYCLDEOL5NGMNGDCN3","short_pith_number":"pith:2SH2B4O5","schema_version":"1.0","canonical_sha256":"d48fa0f1dd0e04b191cbeb4cc698626ee43ebb7bfebebfc3dd50911b7f9d9fd6","source":{"kind":"arxiv","id":"2412.09613","version":1},"attestation_state":"computed","paper":{"title":"PVC: Progressive Visual Token Compression for Unified Image and Video Processing in Large Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chenyu Yang, Hao Tian, Jiahao Wang, Jifeng Dai, Lewei Lu, Weijie Su, Wenhai Wang, Xizhou Zhu, Xuan Dong, Zhe Chen","submitted_at":"2024-12-12T18:59:40Z","abstract_excerpt":"Large Vision-Language Models (VLMs) have been extended to understand both images and videos. Visual token compression is leveraged to reduce the considerable token length of visual inputs. To meet the needs of different tasks, existing high-performance models usually process images and videos separately with different token compression strategies, limiting the capabilities of combining images and videos. To this end, we extend each image into a \"static\" video and introduce a unified token compression strategy called Progressive Visual Token Compression (PVC), where the tokens of each frame are"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.09613","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-12-12T18:59:40Z","cross_cats_sorted":[],"title_canon_sha256":"713c8670f81ee22c00d7e9c946421e9b69ba370e961cd8b30e08d695a451b8cf","abstract_canon_sha256":"9d28ef081141302e3df5a555fd0147e88b89a642fb52ff4adaa12df8d7681aee"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:48:27.207048Z","signature_b64":"AA6h/ObTPYA4xTi3wbErp4wN9TNhCX3I6j4Ks5yW+moTLmeIn2AHWdLe6iFEB9eOk/V4Gt4eywAQRws1pbvgBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d48fa0f1dd0e04b191cbeb4cc698626ee43ebb7bfebebfc3dd50911b7f9d9fd6","last_reissued_at":"2026-07-05T09:48:27.206563Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:48:27.206563Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PVC: Progressive Visual Token Compression for Unified Image and Video Processing in Large Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chenyu Yang, Hao Tian, Jiahao Wang, Jifeng Dai, Lewei Lu, Weijie Su, Wenhai Wang, Xizhou Zhu, Xuan Dong, Zhe Chen","submitted_at":"2024-12-12T18:59:40Z","abstract_excerpt":"Large Vision-Language Models (VLMs) have been extended to understand both images and videos. Visual token compression is leveraged to reduce the considerable token length of visual inputs. To meet the needs of different tasks, existing high-performance models usually process images and videos separately with different token compression strategies, limiting the capabilities of combining images and videos. To this end, we extend each image into a \"static\" video and introduce a unified token compression strategy called Progressive Visual Token Compression (PVC), where the tokens of each frame are"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.09613","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.09613/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.09613","created_at":"2026-07-05T09:48:27.206622+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.09613v1","created_at":"2026-07-05T09:48:27.206622+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.09613","created_at":"2026-07-05T09:48:27.206622+00:00"},{"alias_kind":"pith_short_12","alias_value":"2SH2B4O5BYCL","created_at":"2026-07-05T09:48:27.206622+00:00"},{"alias_kind":"pith_short_16","alias_value":"2SH2B4O5BYCLDEOL","created_at":"2026-07-05T09:48:27.206622+00:00"},{"alias_kind":"pith_short_8","alias_value":"2SH2B4O5","created_at":"2026-07-05T09:48:27.206622+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.01644","citing_title":"CRAFT: Compression via Recursive Adaptive Fusion of Video Tokens for Vision-Language Models","ref_index":15,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2SH2B4O5BYCLDEOL5NGMNGDCN3","json":"https://pith.science/pith/2SH2B4O5BYCLDEOL5NGMNGDCN3.json","graph_json":"https://pith.science/api/pith-number/2SH2B4O5BYCLDEOL5NGMNGDCN3/graph.json","events_json":"https://pith.science/api/pith-number/2SH2B4O5BYCLDEOL5NGMNGDCN3/events.json","paper":"https://pith.science/paper/2SH2B4O5"},"agent_actions":{"view_html":"https://pith.science/pith/2SH2B4O5BYCLDEOL5NGMNGDCN3","download_json":"https://pith.science/pith/2SH2B4O5BYCLDEOL5NGMNGDCN3.json","view_paper":"https://pith.science/paper/2SH2B4O5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.09613&json=true","fetch_graph":"https://pith.science/api/pith-number/2SH2B4O5BYCLDEOL5NGMNGDCN3/graph.json","fetch_events":"https://pith.science/api/pith-number/2SH2B4O5BYCLDEOL5NGMNGDCN3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2SH2B4O5BYCLDEOL5NGMNGDCN3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2SH2B4O5BYCLDEOL5NGMNGDCN3/action/storage_attestation","attest_author":"https://pith.science/pith/2SH2B4O5BYCLDEOL5NGMNGDCN3/action/author_attestation","sign_citation":"https://pith.science/pith/2SH2B4O5BYCLDEOL5NGMNGDCN3/action/citation_signature","submit_replication":"https://pith.science/pith/2SH2B4O5BYCLDEOL5NGMNGDCN3/action/replication_record"}},"created_at":"2026-07-05T09:48:27.206622+00:00","updated_at":"2026-07-05T09:48:27.206622+00:00"}