{"work":{"id":"528509bc-2611-4e7f-a772-ea14d25b6dae","openalex_id":"https://openalex.org/W4306886919","doi":"10.48550/arxiv.2210.09461","arxiv_id":"2210.09461","raw_key":null,"title":"Token Merging: Your ViT But Faster","authors":null,"authors_text":"Daniel Bolya, Cheng-Yang Fu, Xiaoliang Dai, Peizhao Zhang, Christoph Feichtenhofer, Judy Hoffman","year":2022,"venue":"cs.CV","abstract":"We introduce Token Merging (ToMe), a simple method to increase the throughput of existing ViT models without needing to train. ToMe gradually combines similar tokens in a transformer using a general and light-weight matching algorithm that is as fast as pruning while being more accurate. Off-the-shelf, ToMe can 2x the throughput of state-of-the-art ViT-L @ 512 and ViT-H @ 518 models on images and 2.2x the throughput of ViT-L on video with only a 0.2-0.3% accuracy drop in each case. ToMe can also easily be applied during training, improving in practice training speed up to 2x for MAE fine-tuning on video. Training with ToMe further minimizes accuracy drop, leading to 2x the throughput of ViT-B on audio for only a 0.4% mAP drop. Qualitatively, we find that ToMe merges object parts into one token, even over multiple frames of video. Overall, ToMe's accuracy and speed are competitive with state-of-the-art on images, video, and audio.","external_url":"https://arxiv.org/abs/2210.09461","cited_by_count":62,"metadata_source":"pith","metadata_fetched_at":"2026-08-05T02:28:24.338817+00:00","pith_arxiv_id":"2210.09461","created_at":"2026-05-10T06:56:47.577084+00:00","updated_at":"2026-08-05T02:28:24.338817+00:00","title_quality_ok":true,"display_title":"Token Merging: Your ViT But Faster","render_title":"Token Merging: Your ViT But Faster"},"hub":{"state":{"work_id":"528509bc-2611-4e7f-a772-ea14d25b6dae","tier":"hub","tier_reason":"10+ Pith inbound or 1,000+ external citations","pith_inbound_count":82,"external_cited_by_count":62,"distinct_field_count":8,"first_pith_cited_at":"2023-08-16T01:43:41+00:00","last_pith_cited_at":"2026-07-09T01:15:03+00:00","author_build_status":"not_needed","summary_status":"needed","contexts_status":"needed","graph_status":"needed","ask_index_status":"not_needed","reader_status":"not_needed","recognition_status":"not_needed","updated_at":"2026-08-22T16:49:33.074867+00:00","tier_text":"hub"},"tier":"hub","role_counts":[{"context_role":"background","n":9},{"context_role":"method","n":2},{"context_role":"baseline","n":1}],"polarity_counts":[{"context_polarity":"background","n":9},{"context_polarity":"use_method","n":2},{"context_polarity":"baseline","n":1}],"runs":{},"summary":{},"graph":{},"authors":[]}}