{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:HF5CLWADH3GZRPRI647HDZXFOA","short_pith_number":"pith:HF5CLWAD","schema_version":"1.0","canonical_sha256":"397a25d8033ecd98be28f73e71e6e57005e35333688e9a28ab000b3160bd70a2","source":{"kind":"arxiv","id":"2412.09604","version":1},"attestation_state":"computed","paper":{"title":"SynerGen-VL: Towards Synergistic Image Understanding and Generation with Vision Experts and Token Folding","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Changyao Tian, Hao Li, Hongsheng Li, Jie Shao, Jifeng Dai, Jinguo Zhu, Lewei Lu, Wenhan Dou, Xiaogang Wang, Xizhou Zhu, Zhaokai Wang","submitted_at":"2024-12-12T18:59:26Z","abstract_excerpt":"The remarkable success of Large Language Models (LLMs) has extended to the multimodal domain, achieving outstanding performance in image understanding and generation. Recent efforts to develop unified Multimodal Large Language Models (MLLMs) that integrate these capabilities have shown promising results. However, existing approaches often involve complex designs in model architecture or training pipeline, increasing the difficulty of model training and scaling. In this paper, we propose SynerGen-VL, a simple yet powerful encoder-free MLLM capable of both image understanding and generation. To "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.09604","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CV","submitted_at":"2024-12-12T18:59:26Z","cross_cats_sorted":[],"title_canon_sha256":"d1f24d1ed2b1bae35a59ebb4875d86a695aae61e33c116a932d807e3f01b8385","abstract_canon_sha256":"8b3c16a3d7335d14e7496d2da33dc75d654a74e76f9aa853978123eb9431e76d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:48:27.052569Z","signature_b64":"xm1IyL35hbzbUIn/KuZ7i7ASnk1dFHLddY+maw3qSMHyhwSQOFTozxOZ3rJMGKT0QCQAnV9Qxr488PgV3nBkCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"397a25d8033ecd98be28f73e71e6e57005e35333688e9a28ab000b3160bd70a2","last_reissued_at":"2026-07-05T09:48:27.052068Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:48:27.052068Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SynerGen-VL: Towards Synergistic Image Understanding and Generation with Vision Experts and Token Folding","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Changyao Tian, Hao Li, Hongsheng Li, Jie Shao, Jifeng Dai, Jinguo Zhu, Lewei Lu, Wenhan Dou, Xiaogang Wang, Xizhou Zhu, Zhaokai Wang","submitted_at":"2024-12-12T18:59:26Z","abstract_excerpt":"The remarkable success of Large Language Models (LLMs) has extended to the multimodal domain, achieving outstanding performance in image understanding and generation. Recent efforts to develop unified Multimodal Large Language Models (MLLMs) that integrate these capabilities have shown promising results. However, existing approaches often involve complex designs in model architecture or training pipeline, increasing the difficulty of model training and scaling. In this paper, we propose SynerGen-VL, a simple yet powerful encoder-free MLLM capable of both image understanding and generation. To "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.09604","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.09604/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.09604","created_at":"2026-07-05T09:48:27.052131+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.09604v1","created_at":"2026-07-05T09:48:27.052131+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.09604","created_at":"2026-07-05T09:48:27.052131+00:00"},{"alias_kind":"pith_short_12","alias_value":"HF5CLWADH3GZ","created_at":"2026-07-05T09:48:27.052131+00:00"},{"alias_kind":"pith_short_16","alias_value":"HF5CLWADH3GZRPRI","created_at":"2026-07-05T09:48:27.052131+00:00"},{"alias_kind":"pith_short_8","alias_value":"HF5CLWAD","created_at":"2026-07-05T09:48:27.052131+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.14853","citing_title":"Discrimination Is Generation: Unifying Ranking and Retrieval from a Tokenizer Perspective","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2503.07265","citing_title":"WISE: A World Knowledge-Informed Semantic Evaluation for Text-to-Image Generation","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2506.15564","citing_title":"Show-o2: Improved Native Unified Multimodal Models","ref_index":58,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HF5CLWADH3GZRPRI647HDZXFOA","json":"https://pith.science/pith/HF5CLWADH3GZRPRI647HDZXFOA.json","graph_json":"https://pith.science/api/pith-number/HF5CLWADH3GZRPRI647HDZXFOA/graph.json","events_json":"https://pith.science/api/pith-number/HF5CLWADH3GZRPRI647HDZXFOA/events.json","paper":"https://pith.science/paper/HF5CLWAD"},"agent_actions":{"view_html":"https://pith.science/pith/HF5CLWADH3GZRPRI647HDZXFOA","download_json":"https://pith.science/pith/HF5CLWADH3GZRPRI647HDZXFOA.json","view_paper":"https://pith.science/paper/HF5CLWAD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.09604&json=true","fetch_graph":"https://pith.science/api/pith-number/HF5CLWADH3GZRPRI647HDZXFOA/graph.json","fetch_events":"https://pith.science/api/pith-number/HF5CLWADH3GZRPRI647HDZXFOA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HF5CLWADH3GZRPRI647HDZXFOA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HF5CLWADH3GZRPRI647HDZXFOA/action/storage_attestation","attest_author":"https://pith.science/pith/HF5CLWADH3GZRPRI647HDZXFOA/action/author_attestation","sign_citation":"https://pith.science/pith/HF5CLWADH3GZRPRI647HDZXFOA/action/citation_signature","submit_replication":"https://pith.science/pith/HF5CLWADH3GZRPRI647HDZXFOA/action/replication_record"}},"created_at":"2026-07-05T09:48:27.052131+00:00","updated_at":"2026-07-05T09:48:27.052131+00:00"}