{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:YY7DSZM4RGINERXBR7QDZGCQ6J","short_pith_number":"pith:YY7DSZM4","schema_version":"1.0","canonical_sha256":"c63e39659c8990d246e18fe03c9850f26913f83d55f9a957d9af900fc0fe0b49","source":{"kind":"arxiv","id":"2305.12223","version":2},"attestation_state":"computed","paper":{"title":"What Makes for Good Visual Tokenizers for Large Language Models?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Guangzhi Wang, Mohan Kankanhalli, Xiaohan Ding, Ying Shan, Yixiao Ge","submitted_at":"2023-05-20T16:11:26Z","abstract_excerpt":"We empirically investigate proper pre-training methods to build good visual tokenizers, making Large Language Models (LLMs) powerful Multimodal Large Language Models (MLLMs). In our benchmark, which is curated to evaluate MLLMs visual semantic understanding and fine-grained perception capabilities, we discussed different visual tokenizers pre-trained with dominant methods (i.e., DeiT, CLIP, MAE, DINO), and observe that: i) Fully/weakly supervised models capture more semantics than self-supervised models, but the gap is narrowed by scaling up the pre-training dataset. ii) Self-supervised models"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.12223","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-05-20T16:11:26Z","cross_cats_sorted":[],"title_canon_sha256":"78862a73a0ba2a025d8335f55ff35d5e1fd498d8545973048373c665bb25dd95","abstract_canon_sha256":"72c8e56621cf7888198ac124aa141c6af57a3bc0e1cac1bde6b42cb8efd12627"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:12:51.101607Z","signature_b64":"AkaEY0yyp/b2VpY0rZ2ZXlQC+ClSG17Bi3o2Z8WD4+kQ1aevb7KfCYF3UMgdbEOLrpGN925TylgFfqt55QaMAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c63e39659c8990d246e18fe03c9850f26913f83d55f9a957d9af900fc0fe0b49","last_reissued_at":"2026-07-05T06:12:51.101177Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:12:51.101177Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"What Makes for Good Visual Tokenizers for Large Language Models?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Guangzhi Wang, Mohan Kankanhalli, Xiaohan Ding, Ying Shan, Yixiao Ge","submitted_at":"2023-05-20T16:11:26Z","abstract_excerpt":"We empirically investigate proper pre-training methods to build good visual tokenizers, making Large Language Models (LLMs) powerful Multimodal Large Language Models (MLLMs). In our benchmark, which is curated to evaluate MLLMs visual semantic understanding and fine-grained perception capabilities, we discussed different visual tokenizers pre-trained with dominant methods (i.e., DeiT, CLIP, MAE, DINO), and observe that: i) Fully/weakly supervised models capture more semantics than self-supervised models, but the gap is narrowed by scaling up the pre-training dataset. ii) Self-supervised models"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.12223","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.12223/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.12223","created_at":"2026-07-05T06:12:51.101231+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.12223v2","created_at":"2026-07-05T06:12:51.101231+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.12223","created_at":"2026-07-05T06:12:51.101231+00:00"},{"alias_kind":"pith_short_12","alias_value":"YY7DSZM4RGIN","created_at":"2026-07-05T06:12:51.101231+00:00"},{"alias_kind":"pith_short_16","alias_value":"YY7DSZM4RGINERXB","created_at":"2026-07-05T06:12:51.101231+00:00"},{"alias_kind":"pith_short_8","alias_value":"YY7DSZM4","created_at":"2026-07-05T06:12:51.101231+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17030","citing_title":"Qwen-RobotWorld Technical Report: Unifying Embodied World Modeling through Language-Conditioned Video Generation","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2307.16125","citing_title":"SEED-Bench: Benchmarking Multimodal LLMs with Generative Comprehension","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05900","citing_title":"AICA-Bench: Holistically Examining the Capabilities of VLMs in Affective Image Content Analysis","ref_index":39,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YY7DSZM4RGINERXBR7QDZGCQ6J","json":"https://pith.science/pith/YY7DSZM4RGINERXBR7QDZGCQ6J.json","graph_json":"https://pith.science/api/pith-number/YY7DSZM4RGINERXBR7QDZGCQ6J/graph.json","events_json":"https://pith.science/api/pith-number/YY7DSZM4RGINERXBR7QDZGCQ6J/events.json","paper":"https://pith.science/paper/YY7DSZM4"},"agent_actions":{"view_html":"https://pith.science/pith/YY7DSZM4RGINERXBR7QDZGCQ6J","download_json":"https://pith.science/pith/YY7DSZM4RGINERXBR7QDZGCQ6J.json","view_paper":"https://pith.science/paper/YY7DSZM4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.12223&json=true","fetch_graph":"https://pith.science/api/pith-number/YY7DSZM4RGINERXBR7QDZGCQ6J/graph.json","fetch_events":"https://pith.science/api/pith-number/YY7DSZM4RGINERXBR7QDZGCQ6J/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YY7DSZM4RGINERXBR7QDZGCQ6J/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YY7DSZM4RGINERXBR7QDZGCQ6J/action/storage_attestation","attest_author":"https://pith.science/pith/YY7DSZM4RGINERXBR7QDZGCQ6J/action/author_attestation","sign_citation":"https://pith.science/pith/YY7DSZM4RGINERXBR7QDZGCQ6J/action/citation_signature","submit_replication":"https://pith.science/pith/YY7DSZM4RGINERXBR7QDZGCQ6J/action/replication_record"}},"created_at":"2026-07-05T06:12:51.101231+00:00","updated_at":"2026-07-05T06:12:51.101231+00:00"}