{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ZSFM2JJTIUK4WMIJBQ53CBNS7F","short_pith_number":"pith:ZSFM2JJT","schema_version":"1.0","canonical_sha256":"cc8acd25334515cb31090c3bb105b2f954ee67bccf6d3e549dc078e12986a67e","source":{"kind":"arxiv","id":"2309.11499","version":2},"attestation_state":"computed","paper":{"title":"DreamLLM: Synergistic Multimodal Comprehension and Creation","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Chunrui Han, Haoran Wei, Hongyu Zhou, Jianjian Sun, Jinrong Yang, Kaisheng Ma, Liang Zhao, Li Yi, Runpei Dong, Xiangwen Kong, Xiangyu Zhang, Yuang Peng, Zekun Qi, Zheng Ge","submitted_at":"2023-09-20T17:58:05Z","abstract_excerpt":"This paper presents DreamLLM, a learning framework that first achieves versatile Multimodal Large Language Models (MLLMs) empowered with frequently overlooked synergy between multimodal comprehension and creation. DreamLLM operates on two fundamental principles. The first focuses on the generative modeling of both language and image posteriors by direct sampling in the raw multimodal space. This approach circumvents the limitations and information loss inherent to external feature extractors like CLIP, and a more thorough multimodal understanding is obtained. Second, DreamLLM fosters the gener"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.11499","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CV","submitted_at":"2023-09-20T17:58:05Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"d3b6667c847a5fe85293b48bd662f3549f4a21a554606e4ee8b850d253c58712","abstract_canon_sha256":"8dfbbbe0280fc1dadfe93b2717c88408fa33baf4a461020a981ce3bbab48f2ce"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:56:52.606234Z","signature_b64":"QdQSLp1AppDSvP5TKr9dsiJL+xBIKgEVpCQ0N/ue3FEROy2gFFoy+wwuF2VuXjIyJeZYUztKFVheGE5KViZtBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cc8acd25334515cb31090c3bb105b2f954ee67bccf6d3e549dc078e12986a67e","last_reissued_at":"2026-07-05T07:56:52.605726Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:56:52.605726Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DreamLLM: Synergistic Multimodal Comprehension and Creation","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Chunrui Han, Haoran Wei, Hongyu Zhou, Jianjian Sun, Jinrong Yang, Kaisheng Ma, Liang Zhao, Li Yi, Runpei Dong, Xiangwen Kong, Xiangyu Zhang, Yuang Peng, Zekun Qi, Zheng Ge","submitted_at":"2023-09-20T17:58:05Z","abstract_excerpt":"This paper presents DreamLLM, a learning framework that first achieves versatile Multimodal Large Language Models (MLLMs) empowered with frequently overlooked synergy between multimodal comprehension and creation. DreamLLM operates on two fundamental principles. The first focuses on the generative modeling of both language and image posteriors by direct sampling in the raw multimodal space. This approach circumvents the limitations and information loss inherent to external feature extractors like CLIP, and a more thorough multimodal understanding is obtained. Second, DreamLLM fosters the gener"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.11499","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.11499/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.11499","created_at":"2026-07-05T07:56:52.605785+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.11499v2","created_at":"2026-07-05T07:56:52.605785+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.11499","created_at":"2026-07-05T07:56:52.605785+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZSFM2JJTIUK4","created_at":"2026-07-05T07:56:52.605785+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZSFM2JJTIUK4WMIJ","created_at":"2026-07-05T07:56:52.605785+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZSFM2JJT","created_at":"2026-07-05T07:56:52.605785+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":25,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.13289","citing_title":"HYDRA-X: Native Unified Multimodal Models with Holistic Visual Tokenizers","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09132","citing_title":"Vision Language Model Helps Private Information De-Identification in Vision Data","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04264","citing_title":"UniCanvas: A Diffusion-base Unified Model for Text-in-Image Joint Generation","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31603","citing_title":"Lumos-Nexus: Efficient Frequency Bridging with Homogeneous Latent Space for Video Unified Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31326","citing_title":"Bridging Video Understanding and Generation in a Unified Framework","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2505.17726","citing_title":"Slot-MLLM: Object-Centric Visual Tokenization for Multimodal LLM","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2601.21798","citing_title":"CG-MLLM: Captioning and Generating 3D content via Multi-modal Large Language Models","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15876","citing_title":"Unlocking Dense Metric Depth Estimation in VLMs","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15876","citing_title":"Unlocking Dense Metric Depth Estimation in VLMs","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2505.23606","citing_title":"Muddit: Liberating Generation Beyond Text-to-Image with a Unified Discrete Diffusion Model","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2309.15112","citing_title":"InternLM-XComposer: A Vision-Language Large Model for Advanced Text-image Comprehension and Composition","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2401.16420","citing_title":"InternLM-XComposer2: Mastering Free-form Text-Image Composition and Comprehension in Vision-Language Large Model","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2311.07575","citing_title":"SPHINX: The Joint Mixing of Weights, Tasks, and Visual Embeddings for Multi-modal Large Language Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2404.14396","citing_title":"SEED-X: Multimodal Models with Unified Multi-granularity Comprehension and Generation","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2410.13848","citing_title":"Janus: Decoupling Visual Encoding for Unified Multimodal Understanding and Generation","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2311.03079","citing_title":"CogVLM: Visual Expert for Pretrained Language Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2312.14238","citing_title":"InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2403.09631","citing_title":"3D-VLA: A 3D Vision-Language-Action Generative World Model","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2408.11039","citing_title":"Transfusion: Predict the Next Token and Diffuse Images with One Multi-Modal Model","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2505.09568","citing_title":"BLIP3-o: A Family of Fully Open Unified Multimodal Models-Architecture, Training and Dataset","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2409.18869","citing_title":"Emu3: Next-Token Prediction is All You Need","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2501.17811","citing_title":"Janus-Pro: Unified Multimodal Understanding and Generation with Data and Model Scaling","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07753","citing_title":"Symbiotic-MoE: Unlocking the Synergy between Generation and Understanding","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04746","citing_title":"Think in Strokes, Not Pixels: Process-Driven Image Generation via Interleaved Reasoning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18562","citing_title":"AnchorSeg: Language Grounded Query Banks for Reasoning Segmentation","ref_index":193,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZSFM2JJTIUK4WMIJBQ53CBNS7F","json":"https://pith.science/pith/ZSFM2JJTIUK4WMIJBQ53CBNS7F.json","graph_json":"https://pith.science/api/pith-number/ZSFM2JJTIUK4WMIJBQ53CBNS7F/graph.json","events_json":"https://pith.science/api/pith-number/ZSFM2JJTIUK4WMIJBQ53CBNS7F/events.json","paper":"https://pith.science/paper/ZSFM2JJT"},"agent_actions":{"view_html":"https://pith.science/pith/ZSFM2JJTIUK4WMIJBQ53CBNS7F","download_json":"https://pith.science/pith/ZSFM2JJTIUK4WMIJBQ53CBNS7F.json","view_paper":"https://pith.science/paper/ZSFM2JJT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.11499&json=true","fetch_graph":"https://pith.science/api/pith-number/ZSFM2JJTIUK4WMIJBQ53CBNS7F/graph.json","fetch_events":"https://pith.science/api/pith-number/ZSFM2JJTIUK4WMIJBQ53CBNS7F/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZSFM2JJTIUK4WMIJBQ53CBNS7F/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZSFM2JJTIUK4WMIJBQ53CBNS7F/action/storage_attestation","attest_author":"https://pith.science/pith/ZSFM2JJTIUK4WMIJBQ53CBNS7F/action/author_attestation","sign_citation":"https://pith.science/pith/ZSFM2JJTIUK4WMIJBQ53CBNS7F/action/citation_signature","submit_replication":"https://pith.science/pith/ZSFM2JJTIUK4WMIJBQ53CBNS7F/action/replication_record"}},"created_at":"2026-07-05T07:56:52.605785+00:00","updated_at":"2026-07-05T07:56:52.605785+00:00"}