{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:SY2BVIZO6HBW4U4SWAOTKTWBYD","short_pith_number":"pith:SY2BVIZO","schema_version":"1.0","canonical_sha256":"96341aa32ef1c36e5392b01d354ec1c0c5656e5c3f4e6113312416e13053f192","source":{"kind":"arxiv","id":"2405.15232","version":4},"attestation_state":"computed","paper":{"title":"DEEM: Diffusion Models Serve as the Eyes of Large Language Models for Image Perception","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Binyuan Hui, Lei Zhang, Longze Chen, Min Yang, Run Luo, Ting-En Lin, Tongliang Liu, Wanwei He, Xiaobo Xia, Yunshui Li, Zikai Song, Ziqiang Liu","submitted_at":"2024-05-24T05:46:04Z","abstract_excerpt":"The development of large language models (LLMs) has significantly advanced the emergence of large multimodal models (LMMs). While LMMs have achieved tremendous success by promoting the synergy between multimodal comprehension and creation, they often face challenges when confronted with out-of-distribution data, such as which can hardly distinguish orientation, quantity, color, structure, etc. This is primarily due to their reliance on image encoders trained to encode images into task-relevant features, which may lead them to disregard irrelevant details. Delving into the modeling capabilities"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.15232","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-05-24T05:46:04Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"0d7ee0d9461ff1dfadfd989f585da7ebaccda0b642f858ea9a2ed02592b0d1e2","abstract_canon_sha256":"5634791b411f20164b6b44e6fb1d9e2952bff1756602320695d4308911b26880"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:26:37.948923Z","signature_b64":"WvQO4SG5k+iilMOm38Dj9XrRWTvpPoIMwj82opetJidFAPRzKSlOSJeyOWedKFj0frULB8auGZhrJxHZBYcpBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"96341aa32ef1c36e5392b01d354ec1c0c5656e5c3f4e6113312416e13053f192","last_reissued_at":"2026-07-05T10:26:37.948413Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:26:37.948413Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DEEM: Diffusion Models Serve as the Eyes of Large Language Models for Image Perception","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Binyuan Hui, Lei Zhang, Longze Chen, Min Yang, Run Luo, Ting-En Lin, Tongliang Liu, Wanwei He, Xiaobo Xia, Yunshui Li, Zikai Song, Ziqiang Liu","submitted_at":"2024-05-24T05:46:04Z","abstract_excerpt":"The development of large language models (LLMs) has significantly advanced the emergence of large multimodal models (LMMs). While LMMs have achieved tremendous success by promoting the synergy between multimodal comprehension and creation, they often face challenges when confronted with out-of-distribution data, such as which can hardly distinguish orientation, quantity, color, structure, etc. This is primarily due to their reliance on image encoders trained to encode images into task-relevant features, which may lead them to disregard irrelevant details. Delving into the modeling capabilities"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.15232","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.15232/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.15232","created_at":"2026-07-05T10:26:37.948475+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.15232v4","created_at":"2026-07-05T10:26:37.948475+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.15232","created_at":"2026-07-05T10:26:37.948475+00:00"},{"alias_kind":"pith_short_12","alias_value":"SY2BVIZO6HBW","created_at":"2026-07-05T10:26:37.948475+00:00"},{"alias_kind":"pith_short_16","alias_value":"SY2BVIZO6HBW4U4S","created_at":"2026-07-05T10:26:37.948475+00:00"},{"alias_kind":"pith_short_8","alias_value":"SY2BVIZO","created_at":"2026-07-05T10:26:37.948475+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.18714","citing_title":"Semantic Generative Tuning for Unified Multimodal Models","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18714","citing_title":"Semantic Generative Tuning for Unified Multimodal Models","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2409.04429","citing_title":"VILA-U: a Unified Foundation Model Integrating Visual Understanding and Generation","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SY2BVIZO6HBW4U4SWAOTKTWBYD","json":"https://pith.science/pith/SY2BVIZO6HBW4U4SWAOTKTWBYD.json","graph_json":"https://pith.science/api/pith-number/SY2BVIZO6HBW4U4SWAOTKTWBYD/graph.json","events_json":"https://pith.science/api/pith-number/SY2BVIZO6HBW4U4SWAOTKTWBYD/events.json","paper":"https://pith.science/paper/SY2BVIZO"},"agent_actions":{"view_html":"https://pith.science/pith/SY2BVIZO6HBW4U4SWAOTKTWBYD","download_json":"https://pith.science/pith/SY2BVIZO6HBW4U4SWAOTKTWBYD.json","view_paper":"https://pith.science/paper/SY2BVIZO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.15232&json=true","fetch_graph":"https://pith.science/api/pith-number/SY2BVIZO6HBW4U4SWAOTKTWBYD/graph.json","fetch_events":"https://pith.science/api/pith-number/SY2BVIZO6HBW4U4SWAOTKTWBYD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SY2BVIZO6HBW4U4SWAOTKTWBYD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SY2BVIZO6HBW4U4SWAOTKTWBYD/action/storage_attestation","attest_author":"https://pith.science/pith/SY2BVIZO6HBW4U4SWAOTKTWBYD/action/author_attestation","sign_citation":"https://pith.science/pith/SY2BVIZO6HBW4U4SWAOTKTWBYD/action/citation_signature","submit_replication":"https://pith.science/pith/SY2BVIZO6HBW4U4SWAOTKTWBYD/action/replication_record"}},"created_at":"2026-07-05T10:26:37.948475+00:00","updated_at":"2026-07-05T10:26:37.948475+00:00"}