{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:E6NRZPWISU54PGZSHC5D32ILCL","short_pith_number":"pith:E6NRZPWI","schema_version":"1.0","canonical_sha256":"279b1cbec8953bc79b3238ba3de90b12c525e59886e8661b9a8533097fb9f696","source":{"kind":"arxiv","id":"2508.09945","version":1},"attestation_state":"computed","paper":{"title":"VisCodex: Unified Multimodal Code Generation via Merging Vision and Coding Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.CL","authors_text":"Dongdong Zhang, Furu Wei, Lingjie Jiang, Shaohan Huang, Xun Wu, Yixia Li","submitted_at":"2025-08-13T17:00:44Z","abstract_excerpt":"Multimodal large language models (MLLMs) have significantly advanced the integration of visual and textual understanding. However, their ability to generate code from multimodal inputs remains limited. In this work, we introduce VisCodex, a unified framework that seamlessly merges vision and coding language models to empower MLLMs with strong multimodal code generation abilities. Leveraging a task vector-based model merging technique, we integrate a state-of-the-art coding LLM into a strong vision-language backbone, while preserving both visual comprehension and advanced coding skills. To supp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.09945","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-08-13T17:00:44Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"4cb639cdfa14fad5e2271801fd9063c7ab82f0bf19af318f338b18aceda1614e","abstract_canon_sha256":"7e62615e10fb260528dc74e6b8f2ff123aa91c511e4dabf6ccd3572e6b9aa6e4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:53:24.933234Z","signature_b64":"tvJUgzXlEoBuOxe3MA3ZoX51dCkJZidvh/6dNFJjnPZx1f3gTnueGBUNSkglhymXdXIShOu8vTiTFaKNE8QIBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"279b1cbec8953bc79b3238ba3de90b12c525e59886e8661b9a8533097fb9f696","last_reissued_at":"2026-07-05T11:53:24.932732Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:53:24.932732Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VisCodex: Unified Multimodal Code Generation via Merging Vision and Coding Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.CL","authors_text":"Dongdong Zhang, Furu Wei, Lingjie Jiang, Shaohan Huang, Xun Wu, Yixia Li","submitted_at":"2025-08-13T17:00:44Z","abstract_excerpt":"Multimodal large language models (MLLMs) have significantly advanced the integration of visual and textual understanding. However, their ability to generate code from multimodal inputs remains limited. In this work, we introduce VisCodex, a unified framework that seamlessly merges vision and coding language models to empower MLLMs with strong multimodal code generation abilities. Leveraging a task vector-based model merging technique, we integrate a state-of-the-art coding LLM into a strong vision-language backbone, while preserving both visual comprehension and advanced coding skills. To supp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.09945","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.09945/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.09945","created_at":"2026-07-05T11:53:24.932794+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.09945v1","created_at":"2026-07-05T11:53:24.932794+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.09945","created_at":"2026-07-05T11:53:24.932794+00:00"},{"alias_kind":"pith_short_12","alias_value":"E6NRZPWISU54","created_at":"2026-07-05T11:53:24.932794+00:00"},{"alias_kind":"pith_short_16","alias_value":"E6NRZPWISU54PGZS","created_at":"2026-07-05T11:53:24.932794+00:00"},{"alias_kind":"pith_short_8","alias_value":"E6NRZPWI","created_at":"2026-07-05T11:53:24.932794+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.15932","citing_title":"Beyond NL2Code: A Structured Survey of Multimodal Code Intelligence","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03378","citing_title":"Neural Change Prediction: Relating Software Changes to Their Effects and Vice Versa","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2603.13224","citing_title":"Visual-ERM: Reward Modeling for Visual Equivalence","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22192","citing_title":"CharTide: Data-Centric Chart-to-Code Generation via Tri-Perspective Tuning and Inquiry-Driven Evolution","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/E6NRZPWISU54PGZSHC5D32ILCL","json":"https://pith.science/pith/E6NRZPWISU54PGZSHC5D32ILCL.json","graph_json":"https://pith.science/api/pith-number/E6NRZPWISU54PGZSHC5D32ILCL/graph.json","events_json":"https://pith.science/api/pith-number/E6NRZPWISU54PGZSHC5D32ILCL/events.json","paper":"https://pith.science/paper/E6NRZPWI"},"agent_actions":{"view_html":"https://pith.science/pith/E6NRZPWISU54PGZSHC5D32ILCL","download_json":"https://pith.science/pith/E6NRZPWISU54PGZSHC5D32ILCL.json","view_paper":"https://pith.science/paper/E6NRZPWI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.09945&json=true","fetch_graph":"https://pith.science/api/pith-number/E6NRZPWISU54PGZSHC5D32ILCL/graph.json","fetch_events":"https://pith.science/api/pith-number/E6NRZPWISU54PGZSHC5D32ILCL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/E6NRZPWISU54PGZSHC5D32ILCL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/E6NRZPWISU54PGZSHC5D32ILCL/action/storage_attestation","attest_author":"https://pith.science/pith/E6NRZPWISU54PGZSHC5D32ILCL/action/author_attestation","sign_citation":"https://pith.science/pith/E6NRZPWISU54PGZSHC5D32ILCL/action/citation_signature","submit_replication":"https://pith.science/pith/E6NRZPWISU54PGZSHC5D32ILCL/action/replication_record"}},"created_at":"2026-07-05T11:53:24.932794+00:00","updated_at":"2026-07-05T11:53:24.932794+00:00"}