{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FVZZXBKPSOB6L2HPEVGRGO6HWK","short_pith_number":"pith:FVZZXBKP","schema_version":"1.0","canonical_sha256":"2d739b854f9383e5e8ef254d133bc7b293dff195bd56684ac8a779c1374ad7b9","source":{"kind":"arxiv","id":"2506.03990","version":1},"attestation_state":"computed","paper":{"title":"DynTok: Dynamic Compression of Visual Tokens for Efficient and Effective Video Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Fuzheng Zhang, Hongzhi Zhang, Jingyuan Zhang, Qi Wang, Xingguang Ji","submitted_at":"2025-06-04T14:17:42Z","abstract_excerpt":"Typical video modeling methods, such as LLava, represent videos as sequences of visual tokens, which are then processed by the LLM backbone for effective video understanding. However, this approach leads to a massive number of visual tokens, especially for long videos. A practical solution is to first extract relevant visual information from the large visual context before feeding it into the LLM backbone, thereby reducing computational overhead. In this work, we introduce DynTok, a novel \\textbf{Dyn}amic video \\textbf{Tok}en compression strategy. DynTok adaptively splits visual tokens into gr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.03990","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-04T14:17:42Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"42a99198fe68086cb4abd7ce72b9824ce2b504c32462237fd8ec3a66414501d5","abstract_canon_sha256":"97013a5fa26db87dddd80b80671de806730c1e7369427fb15051c1c535f46669"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:15:55.003516Z","signature_b64":"4dOh+xYtiwZm9Ttw1LSKn6P8hAZISs7WgWCFpkRpapM0Dga6cUgWEbES/jSgZz/vTKGDQpsyPRo7j2B6ujBLAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2d739b854f9383e5e8ef254d133bc7b293dff195bd56684ac8a779c1374ad7b9","last_reissued_at":"2026-07-05T11:15:55.003044Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:15:55.003044Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DynTok: Dynamic Compression of Visual Tokens for Efficient and Effective Video Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Fuzheng Zhang, Hongzhi Zhang, Jingyuan Zhang, Qi Wang, Xingguang Ji","submitted_at":"2025-06-04T14:17:42Z","abstract_excerpt":"Typical video modeling methods, such as LLava, represent videos as sequences of visual tokens, which are then processed by the LLM backbone for effective video understanding. However, this approach leads to a massive number of visual tokens, especially for long videos. A practical solution is to first extract relevant visual information from the large visual context before feeding it into the LLM backbone, thereby reducing computational overhead. In this work, we introduce DynTok, a novel \\textbf{Dyn}amic video \\textbf{Tok}en compression strategy. DynTok adaptively splits visual tokens into gr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.03990","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.03990/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.03990","created_at":"2026-07-05T11:15:55.003102+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.03990v1","created_at":"2026-07-05T11:15:55.003102+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.03990","created_at":"2026-07-05T11:15:55.003102+00:00"},{"alias_kind":"pith_short_12","alias_value":"FVZZXBKPSOB6","created_at":"2026-07-05T11:15:55.003102+00:00"},{"alias_kind":"pith_short_16","alias_value":"FVZZXBKPSOB6L2HP","created_at":"2026-07-05T11:15:55.003102+00:00"},{"alias_kind":"pith_short_8","alias_value":"FVZZXBKP","created_at":"2026-07-05T11:15:55.003102+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19849","citing_title":"ViCoStream: Streaming VideoLLMs Can Run Beyond 100 FPS with Stage-Wise Coordinated Inference","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31247","citing_title":"FlexiSLM: A Dynamic and Controllable Frame Rate Spoken Language Model","ref_index":248,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FVZZXBKPSOB6L2HPEVGRGO6HWK","json":"https://pith.science/pith/FVZZXBKPSOB6L2HPEVGRGO6HWK.json","graph_json":"https://pith.science/api/pith-number/FVZZXBKPSOB6L2HPEVGRGO6HWK/graph.json","events_json":"https://pith.science/api/pith-number/FVZZXBKPSOB6L2HPEVGRGO6HWK/events.json","paper":"https://pith.science/paper/FVZZXBKP"},"agent_actions":{"view_html":"https://pith.science/pith/FVZZXBKPSOB6L2HPEVGRGO6HWK","download_json":"https://pith.science/pith/FVZZXBKPSOB6L2HPEVGRGO6HWK.json","view_paper":"https://pith.science/paper/FVZZXBKP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.03990&json=true","fetch_graph":"https://pith.science/api/pith-number/FVZZXBKPSOB6L2HPEVGRGO6HWK/graph.json","fetch_events":"https://pith.science/api/pith-number/FVZZXBKPSOB6L2HPEVGRGO6HWK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FVZZXBKPSOB6L2HPEVGRGO6HWK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FVZZXBKPSOB6L2HPEVGRGO6HWK/action/storage_attestation","attest_author":"https://pith.science/pith/FVZZXBKPSOB6L2HPEVGRGO6HWK/action/author_attestation","sign_citation":"https://pith.science/pith/FVZZXBKPSOB6L2HPEVGRGO6HWK/action/citation_signature","submit_replication":"https://pith.science/pith/FVZZXBKPSOB6L2HPEVGRGO6HWK/action/replication_record"}},"created_at":"2026-07-05T11:15:55.003102+00:00","updated_at":"2026-07-05T11:15:55.003102+00:00"}