{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:AXQWGILUX7XM2MRKLVEOSLIMCT","short_pith_number":"pith:AXQWGILU","schema_version":"1.0","canonical_sha256":"05e1632174bfeecd322a5d48e92d0c14f191635cd18c6d1dd1a88a3c47d1f389","source":{"kind":"arxiv","id":"2410.08368","version":2},"attestation_state":"computed","paper":{"title":"ElasticTok: Adaptive Tokenization for Image and Video","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Aleksandra Faust, Hao Liu, Matei Zaharia, Pieter Abbeel, Volodymyr Mnih, Wilson Yan","submitted_at":"2024-10-10T20:54:15Z","abstract_excerpt":"Efficient video tokenization remains a key bottleneck in learning general purpose vision models that are capable of processing long video sequences. Prevailing approaches are restricted to encoding videos to a fixed number of tokens, where too few tokens will result in overly lossy encodings, and too many tokens will result in prohibitively long sequence lengths. In this work, we introduce ElasticTok, a method that conditions on prior frames to adaptively encode a frame into a variable number of tokens. To enable this in a computationally scalable way, we propose a masking technique that drops"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.08368","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-10-10T20:54:15Z","cross_cats_sorted":[],"title_canon_sha256":"e047448763e190bcf1e619c034238d25064d54a5ca2269d9892107f70731bc98","abstract_canon_sha256":"f10d8c7b79184bc0a623e8e9c978e4a2d192e24c52facd7f3209f48bec2b7fb8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:08:22.641183Z","signature_b64":"UelsVqW5YG3+mchR8mEZIg2QPxTCkUjgm7kD15byoOSNS5oTt7uzvqIqdoMmtdOD3IiyCSFeOXKxQ+GlVHttBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"05e1632174bfeecd322a5d48e92d0c14f191635cd18c6d1dd1a88a3c47d1f389","last_reissued_at":"2026-07-05T10:08:22.640665Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:08:22.640665Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ElasticTok: Adaptive Tokenization for Image and Video","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Aleksandra Faust, Hao Liu, Matei Zaharia, Pieter Abbeel, Volodymyr Mnih, Wilson Yan","submitted_at":"2024-10-10T20:54:15Z","abstract_excerpt":"Efficient video tokenization remains a key bottleneck in learning general purpose vision models that are capable of processing long video sequences. Prevailing approaches are restricted to encoding videos to a fixed number of tokens, where too few tokens will result in overly lossy encodings, and too many tokens will result in prohibitively long sequence lengths. In this work, we introduce ElasticTok, a method that conditions on prior frames to adaptively encode a frame into a variable number of tokens. To enable this in a computationally scalable way, we propose a masking technique that drops"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.08368","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.08368/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.08368","created_at":"2026-07-05T10:08:22.640732+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.08368v2","created_at":"2026-07-05T10:08:22.640732+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.08368","created_at":"2026-07-05T10:08:22.640732+00:00"},{"alias_kind":"pith_short_12","alias_value":"AXQWGILUX7XM","created_at":"2026-07-05T10:08:22.640732+00:00"},{"alias_kind":"pith_short_16","alias_value":"AXQWGILUX7XM2MRK","created_at":"2026-07-05T10:08:22.640732+00:00"},{"alias_kind":"pith_short_8","alias_value":"AXQWGILU","created_at":"2026-07-05T10:08:22.640732+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17590","citing_title":"TivTok: Broadcasting Time-Invariant Tokens for Scalable Video Tokenization","ref_index":93,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07185","citing_title":"AdaTok: Self-Budgeting Image Tokenization with Quality-Preserving Dynamic Tokens","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06158","citing_title":"Adaptive Tokenisation Via Temporal Redundancy Masking And Latent Inpainting","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04461","citing_title":"ChannelTok: Efficient Flexible-Length Vision Tokenization","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03578","citing_title":"Diffusing in the Right Space: A Systematic Study of Latent Diffusability","ref_index":90,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27696","citing_title":"Structure over Pixels: Learning Variable-Length Visual Programs","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22678","citing_title":"Swift Sampling: Selecting Temporal Surprises via Taylor Series","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2510.18091","citing_title":"Accelerating Vision Transformers with Adaptive Patch Sizes","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2603.06351","citing_title":"DC-DiT: Adaptive Compute and Elastic Inference for Visual Generation via Dynamic Chunking","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2501.09747","citing_title":"FAST: Efficient Action Tokenization for Vision-Language-Action Models","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09168","citing_title":"ELT: Elastic Looped Transformers for Visual Generation","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07915","citing_title":"What Matters for Diffusion-Friendly Latent Manifold? Prior-Aligned Autoencoders for Latent Diffusion","ref_index":94,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AXQWGILUX7XM2MRKLVEOSLIMCT","json":"https://pith.science/pith/AXQWGILUX7XM2MRKLVEOSLIMCT.json","graph_json":"https://pith.science/api/pith-number/AXQWGILUX7XM2MRKLVEOSLIMCT/graph.json","events_json":"https://pith.science/api/pith-number/AXQWGILUX7XM2MRKLVEOSLIMCT/events.json","paper":"https://pith.science/paper/AXQWGILU"},"agent_actions":{"view_html":"https://pith.science/pith/AXQWGILUX7XM2MRKLVEOSLIMCT","download_json":"https://pith.science/pith/AXQWGILUX7XM2MRKLVEOSLIMCT.json","view_paper":"https://pith.science/paper/AXQWGILU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.08368&json=true","fetch_graph":"https://pith.science/api/pith-number/AXQWGILUX7XM2MRKLVEOSLIMCT/graph.json","fetch_events":"https://pith.science/api/pith-number/AXQWGILUX7XM2MRKLVEOSLIMCT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AXQWGILUX7XM2MRKLVEOSLIMCT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AXQWGILUX7XM2MRKLVEOSLIMCT/action/storage_attestation","attest_author":"https://pith.science/pith/AXQWGILUX7XM2MRKLVEOSLIMCT/action/author_attestation","sign_citation":"https://pith.science/pith/AXQWGILUX7XM2MRKLVEOSLIMCT/action/citation_signature","submit_replication":"https://pith.science/pith/AXQWGILUX7XM2MRKLVEOSLIMCT/action/replication_record"}},"created_at":"2026-07-05T10:08:22.640732+00:00","updated_at":"2026-07-05T10:08:22.640732+00:00"}