{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:PSZOHUPKCGNTNMHYE72UNIC4N7","short_pith_number":"pith:PSZOHUPK","schema_version":"1.0","canonical_sha256":"7cb2e3d1ea119b36b0f827f546a05c6fc524d5a527da3205e44137c5666ff332","source":{"kind":"arxiv","id":"2504.01934","version":2},"attestation_state":"computed","paper":{"title":"ILLUME+: Illuminating Unified MLLM with Dual Visual Tokenization and Diffusion Refinement","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chunwei Wang, Guansong Lu, Hang Xu, Hengshuang Zhao, Jianhua Han, Junwei Yang, Lanqing Hong, Lu Hou, Runhui Huang, Wei Zhang, Yunlong Yuan","submitted_at":"2025-04-02T17:45:00Z","abstract_excerpt":"We present ILLUME+ that leverages dual visual tokenization and a diffusion decoder to improve both deep semantic understanding and high-fidelity image generation. Existing unified models have struggled to simultaneously handle the three fundamental capabilities in a unified model: understanding, generation, and editing. Models like Chameleon and EMU3 utilize VQGAN for image discretization, due to the lack of deep semantic interaction, they lag behind specialist models like LLaVA in visual understanding tasks. To mitigate this, LaViT and ILLUME employ semantic encoders for tokenization, but the"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.01934","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-04-02T17:45:00Z","cross_cats_sorted":[],"title_canon_sha256":"4d1679247755529f7636bf092ea19510b68374ac733e5c6c701e76e2e7bf92d8","abstract_canon_sha256":"30024a79656c4c731b30ac64ebcee641e5d831fdb760a61a4c2850754127f915"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:43:47.879886Z","signature_b64":"7kzczOtuq5IU+iALPPRdjtWb1yLvZl/Jvr8AftYD5dpVvqXUYHF4sHPI1dhyuIDzFVvjvy7Np4c5FQ85e0+iBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7cb2e3d1ea119b36b0f827f546a05c6fc524d5a527da3205e44137c5666ff332","last_reissued_at":"2026-07-05T10:43:47.879409Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:43:47.879409Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ILLUME+: Illuminating Unified MLLM with Dual Visual Tokenization and Diffusion Refinement","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chunwei Wang, Guansong Lu, Hang Xu, Hengshuang Zhao, Jianhua Han, Junwei Yang, Lanqing Hong, Lu Hou, Runhui Huang, Wei Zhang, Yunlong Yuan","submitted_at":"2025-04-02T17:45:00Z","abstract_excerpt":"We present ILLUME+ that leverages dual visual tokenization and a diffusion decoder to improve both deep semantic understanding and high-fidelity image generation. Existing unified models have struggled to simultaneously handle the three fundamental capabilities in a unified model: understanding, generation, and editing. Models like Chameleon and EMU3 utilize VQGAN for image discretization, due to the lack of deep semantic interaction, they lag behind specialist models like LLaVA in visual understanding tasks. To mitigate this, LaViT and ILLUME employ semantic encoders for tokenization, but the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.01934","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.01934/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.01934","created_at":"2026-07-05T10:43:47.879472+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.01934v2","created_at":"2026-07-05T10:43:47.879472+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.01934","created_at":"2026-07-05T10:43:47.879472+00:00"},{"alias_kind":"pith_short_12","alias_value":"PSZOHUPKCGNT","created_at":"2026-07-05T10:43:47.879472+00:00"},{"alias_kind":"pith_short_16","alias_value":"PSZOHUPKCGNTNMHY","created_at":"2026-07-05T10:43:47.879472+00:00"},{"alias_kind":"pith_short_8","alias_value":"PSZOHUPK","created_at":"2026-07-05T10:43:47.879472+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24333","citing_title":"UniTranslator: A Unified Multi-modal Framework for End-to-end In-Image Machine Translation","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26984","citing_title":"Unison: Benchmarking Unified Multimodal Models via Synergistic Understanding and Generation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22913","citing_title":"Intend, Reflect, Refine: An Adaptive Multimodal Reflection Framework for Autonomous Driving","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23041","citing_title":"SPAR: Semantic-Pixel Self-Alignment and Adaptive Routing for Unified Multimodal Models","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23041","citing_title":"SPAR: Semantic-Pixel Self-Alignment and Adaptive Routing for Unified Multimodal Models","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17296","citing_title":"Pareto LoRA: Mitigating Modality Imbalance in Unified Multimodal Models via Pareto-Optimal Gradient Integration","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02402","citing_title":"Show Me Examples: Inferring Visual Concepts from Image Sets","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13289","citing_title":"HYDRA-X: Native Unified Multimodal Models with Holistic Visual Tokenizers","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20795","citing_title":"What Semantics Survive the Connector? Diagnosing VLM-to-DiT Alignment in Video Editing","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30054","citing_title":"Illuminating Unified Multimodal Model for Free-form Interleaved Text-Image Generation","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20795","citing_title":"What Semantics Survive the Connector? Diagnosing VLM-to-DiT Alignment in Video Editing","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12500","citing_title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2506.15564","citing_title":"Show-o2: Improved Native Unified Multimodal Models","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08724","citing_title":"SynerMedGen: Synergizing Medical Multimodal Understanding with Generation via Task Alignment","ref_index":60,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PSZOHUPKCGNTNMHYE72UNIC4N7","json":"https://pith.science/pith/PSZOHUPKCGNTNMHYE72UNIC4N7.json","graph_json":"https://pith.science/api/pith-number/PSZOHUPKCGNTNMHYE72UNIC4N7/graph.json","events_json":"https://pith.science/api/pith-number/PSZOHUPKCGNTNMHYE72UNIC4N7/events.json","paper":"https://pith.science/paper/PSZOHUPK"},"agent_actions":{"view_html":"https://pith.science/pith/PSZOHUPKCGNTNMHYE72UNIC4N7","download_json":"https://pith.science/pith/PSZOHUPKCGNTNMHYE72UNIC4N7.json","view_paper":"https://pith.science/paper/PSZOHUPK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.01934&json=true","fetch_graph":"https://pith.science/api/pith-number/PSZOHUPKCGNTNMHYE72UNIC4N7/graph.json","fetch_events":"https://pith.science/api/pith-number/PSZOHUPKCGNTNMHYE72UNIC4N7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PSZOHUPKCGNTNMHYE72UNIC4N7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PSZOHUPKCGNTNMHYE72UNIC4N7/action/storage_attestation","attest_author":"https://pith.science/pith/PSZOHUPKCGNTNMHYE72UNIC4N7/action/author_attestation","sign_citation":"https://pith.science/pith/PSZOHUPKCGNTNMHYE72UNIC4N7/action/citation_signature","submit_replication":"https://pith.science/pith/PSZOHUPKCGNTNMHYE72UNIC4N7/action/replication_record"}},"created_at":"2026-07-05T10:43:47.879472+00:00","updated_at":"2026-07-05T10:43:47.879472+00:00"}