{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:7HDKIP4UT74IRIKBSLK3P4DSZP","short_pith_number":"pith:7HDKIP4U","schema_version":"1.0","canonical_sha256":"f9c6a43f949ff888a14192d5b7f072cbe1fa0079bf2a57cd83955ce6fae102d2","source":{"kind":"arxiv","id":"2406.05127","version":4},"attestation_state":"computed","paper":{"title":"Towards Semantic Equivalence of Tokenization in Multimodal LLM","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hanwang Zhang, Hao Fei, Jiayi Ji, Shengqiong Wu, Shuicheng Yan, Tat-Seng Chua, Xiangtai Li","submitted_at":"2024-06-07T17:55:43Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have demonstrated exceptional capabilities in processing vision-language tasks. One of the crux of MLLMs lies in vision tokenization, which involves efficiently transforming input visual signals into feature representations that are most beneficial for LLMs. However, existing vision tokenizers, essential for semantic alignment between vision and language, remain problematic. Existing methods aggressively fragment visual input, corrupting the visual semantic integrity. To address this, this paper proposes a novel dynamic Semantic-Equivalent Vision Tokeni"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.05127","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-06-07T17:55:43Z","cross_cats_sorted":[],"title_canon_sha256":"33508e4b7423ec290ab189ec4600a6c89b670610db156de958d7206e2fdb9ed4","abstract_canon_sha256":"17e00bc74b44100b3e3a439d6e0c8f3de85a2fcff795439562e4df4524bf988a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:19:55.351270Z","signature_b64":"w/srxaSto0lIlEa7EzQwZjzj1wk8FeKfFpRn+IOTV3q3kq5jSJw7KOHj1wz21bLWTjX6kB69WeeyPuIROTc1Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f9c6a43f949ff888a14192d5b7f072cbe1fa0079bf2a57cd83955ce6fae102d2","last_reissued_at":"2026-07-05T10:19:55.350865Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:19:55.350865Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Semantic Equivalence of Tokenization in Multimodal LLM","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hanwang Zhang, Hao Fei, Jiayi Ji, Shengqiong Wu, Shuicheng Yan, Tat-Seng Chua, Xiangtai Li","submitted_at":"2024-06-07T17:55:43Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have demonstrated exceptional capabilities in processing vision-language tasks. One of the crux of MLLMs lies in vision tokenization, which involves efficiently transforming input visual signals into feature representations that are most beneficial for LLMs. However, existing vision tokenizers, essential for semantic alignment between vision and language, remain problematic. Existing methods aggressively fragment visual input, corrupting the visual semantic integrity. To address this, this paper proposes a novel dynamic Semantic-Equivalent Vision Tokeni"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.05127","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.05127/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.05127","created_at":"2026-07-05T10:19:55.350918+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.05127v4","created_at":"2026-07-05T10:19:55.350918+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.05127","created_at":"2026-07-05T10:19:55.350918+00:00"},{"alias_kind":"pith_short_12","alias_value":"7HDKIP4UT74I","created_at":"2026-07-05T10:19:55.350918+00:00"},{"alias_kind":"pith_short_16","alias_value":"7HDKIP4UT74IRIKB","created_at":"2026-07-05T10:19:55.350918+00:00"},{"alias_kind":"pith_short_8","alias_value":"7HDKIP4U","created_at":"2026-07-05T10:19:55.350918+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2505.17726","citing_title":"Slot-MLLM: Object-Centric Visual Tokenization for Multimodal LLM","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17954","citing_title":"A More Word-like Image Tokenization for MLLMs","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2603.03944","citing_title":"SCP: Spatial Causal Prediction in Video","ref_index":50,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7HDKIP4UT74IRIKBSLK3P4DSZP","json":"https://pith.science/pith/7HDKIP4UT74IRIKBSLK3P4DSZP.json","graph_json":"https://pith.science/api/pith-number/7HDKIP4UT74IRIKBSLK3P4DSZP/graph.json","events_json":"https://pith.science/api/pith-number/7HDKIP4UT74IRIKBSLK3P4DSZP/events.json","paper":"https://pith.science/paper/7HDKIP4U"},"agent_actions":{"view_html":"https://pith.science/pith/7HDKIP4UT74IRIKBSLK3P4DSZP","download_json":"https://pith.science/pith/7HDKIP4UT74IRIKBSLK3P4DSZP.json","view_paper":"https://pith.science/paper/7HDKIP4U","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.05127&json=true","fetch_graph":"https://pith.science/api/pith-number/7HDKIP4UT74IRIKBSLK3P4DSZP/graph.json","fetch_events":"https://pith.science/api/pith-number/7HDKIP4UT74IRIKBSLK3P4DSZP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7HDKIP4UT74IRIKBSLK3P4DSZP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7HDKIP4UT74IRIKBSLK3P4DSZP/action/storage_attestation","attest_author":"https://pith.science/pith/7HDKIP4UT74IRIKBSLK3P4DSZP/action/author_attestation","sign_citation":"https://pith.science/pith/7HDKIP4UT74IRIKBSLK3P4DSZP/action/citation_signature","submit_replication":"https://pith.science/pith/7HDKIP4UT74IRIKBSLK3P4DSZP/action/replication_record"}},"created_at":"2026-07-05T10:19:55.350918+00:00","updated_at":"2026-07-05T10:19:55.350918+00:00"}