{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:72JWBN34NGDST6NS4X4YRLCVZH","short_pith_number":"pith:72JWBN34","schema_version":"1.0","canonical_sha256":"fe9360b77c698729f9b2e5f988ac55c9f827ff855e673c7cbf36869a85b87146","source":{"kind":"arxiv","id":"2402.12226","version":5},"attestation_state":"computed","paper":{"title":"AnyGPT: Unified Multimodal LLM with Discrete Sequence Modeling","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG"],"primary_cat":"cs.CL","authors_text":"Dong Zhang, Ge Zhang, Hang Yan, Jiasheng Ye, Jie Fu, Junqi Dai, Jun Zhan, Linyang Li, Ruibin Yuan, Tao Gui, Tianxiang Sun, Xin Zhang, Xipeng Qiu, Yu-Gang Jiang, Yunhua Zhou, Zhigeng Liu","submitted_at":"2024-02-19T15:33:10Z","abstract_excerpt":"We introduce AnyGPT, an any-to-any multimodal language model that utilizes discrete representations for the unified processing of various modalities, including speech, text, images, and music. AnyGPT can be trained stably without any alterations to the current large language model (LLM) architecture or training paradigms. Instead, it relies exclusively on data-level preprocessing, facilitating the seamless integration of new modalities into LLMs, akin to the incorporation of new languages. We build a multimodal text-centric dataset for multimodal alignment pre-training. Utilizing generative mo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.12226","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-02-19T15:33:10Z","cross_cats_sorted":["cs.AI","cs.CV","cs.LG"],"title_canon_sha256":"1feb306d48308873e1370d24b96d409d83cb81c65c3d221c491b2d371e43891c","abstract_canon_sha256":"1c41e3946a7dbcc5034d9d947b54a6a92a3c4c375cee096ed7770def7ba71de1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:05:59.169470Z","signature_b64":"0LkSrXvoWGx7r/qIWCpJN6EOOrjnKgqX4rCDtmI1aM+lE1GX1DqwuMGQowUHboAZgnR3Q9ok1f7hIw0RDrhBAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fe9360b77c698729f9b2e5f988ac55c9f827ff855e673c7cbf36869a85b87146","last_reissued_at":"2026-07-05T12:05:59.168877Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:05:59.168877Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AnyGPT: Unified Multimodal LLM with Discrete Sequence Modeling","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG"],"primary_cat":"cs.CL","authors_text":"Dong Zhang, Ge Zhang, Hang Yan, Jiasheng Ye, Jie Fu, Junqi Dai, Jun Zhan, Linyang Li, Ruibin Yuan, Tao Gui, Tianxiang Sun, Xin Zhang, Xipeng Qiu, Yu-Gang Jiang, Yunhua Zhou, Zhigeng Liu","submitted_at":"2024-02-19T15:33:10Z","abstract_excerpt":"We introduce AnyGPT, an any-to-any multimodal language model that utilizes discrete representations for the unified processing of various modalities, including speech, text, images, and music. AnyGPT can be trained stably without any alterations to the current large language model (LLM) architecture or training paradigms. Instead, it relies exclusively on data-level preprocessing, facilitating the seamless integration of new modalities into LLMs, akin to the incorporation of new languages. We build a multimodal text-centric dataset for multimodal alignment pre-training. Utilizing generative mo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.12226","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.12226/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.12226","created_at":"2026-07-05T12:05:59.168955+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.12226v5","created_at":"2026-07-05T12:05:59.168955+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.12226","created_at":"2026-07-05T12:05:59.168955+00:00"},{"alias_kind":"pith_short_12","alias_value":"72JWBN34NGDS","created_at":"2026-07-05T12:05:59.168955+00:00"},{"alias_kind":"pith_short_16","alias_value":"72JWBN34NGDST6NS","created_at":"2026-07-05T12:05:59.168955+00:00"},{"alias_kind":"pith_short_8","alias_value":"72JWBN34","created_at":"2026-07-05T12:05:59.168955+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06649","citing_title":"POPS: Recovering Unlearned Multi-Modality Knowledge in MLLMs with Prompt-Optimized Parameter Shaking","ref_index":18,"is_internal_anchor":true},{"citing_arxiv_id":"2606.22873","citing_title":"SingGuard: A Policy-Adaptive Multimodal LLM Guardrail with Dynamic Reasoning","ref_index":300,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12940","citing_title":"Self-Guidance: Enhancing Neural Codecs via Decoder Manifold Alignment","ref_index":140,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07643","citing_title":"AVI-Bench: Toward Human-like Audio-Visual Intelligence of Omni-MLLMs","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22873","citing_title":"SingGuard: A Policy-Adaptive Multimodal LLM Guardrail with Dynamic Reasoning","ref_index":299,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25343","citing_title":"Toward Native Multimodal Modeling: A Roadmap","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2512.20901","citing_title":"Benchmarking and Enhancing VLM for Compressed Image Understanding","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2512.01537","citing_title":"Two-Dimensional Quantization for Geometry-Aware Audio Coding","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2409.07825","citing_title":"Deep Multimodal Learning with Missing Modality: A Survey","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2501.01957","citing_title":"VITA-1.5: Towards GPT-4o Level Real-Time Vision and Speech Interaction","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2403.18814","citing_title":"Mini-Gemini: Mining the Potential of Multi-modality Vision Language Models","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2512.14234","citing_title":"ViBES: A Conversational Agent with Behaviorally-Intelligent 3D Virtual Body","ref_index":130,"is_internal_anchor":false},{"citing_arxiv_id":"2306.13549","citing_title":"A Survey on Multimodal Large Language Models","ref_index":149,"is_internal_anchor":false},{"citing_arxiv_id":"2409.04429","citing_title":"VILA-U: a Unified Foundation Model Integrating Visual Understanding and Generation","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2503.12605","citing_title":"Multimodal Chain-of-Thought Reasoning: A Comprehensive Survey","ref_index":213,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11605","citing_title":"Keep What Audio Cannot Say: Context-Preserving Token Pruning for Omni-LLMs","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21921","citing_title":"Context Unrolling in Omni Models","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08125","citing_title":"PolySLGen: Online Multimodal Speaking-Listening Reaction Generation in Polyadic Interaction","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2503.20215","citing_title":"Qwen2.5-Omni Technical Report","ref_index":42,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/72JWBN34NGDST6NS4X4YRLCVZH","json":"https://pith.science/pith/72JWBN34NGDST6NS4X4YRLCVZH.json","graph_json":"https://pith.science/api/pith-number/72JWBN34NGDST6NS4X4YRLCVZH/graph.json","events_json":"https://pith.science/api/pith-number/72JWBN34NGDST6NS4X4YRLCVZH/events.json","paper":"https://pith.science/paper/72JWBN34"},"agent_actions":{"view_html":"https://pith.science/pith/72JWBN34NGDST6NS4X4YRLCVZH","download_json":"https://pith.science/pith/72JWBN34NGDST6NS4X4YRLCVZH.json","view_paper":"https://pith.science/paper/72JWBN34","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.12226&json=true","fetch_graph":"https://pith.science/api/pith-number/72JWBN34NGDST6NS4X4YRLCVZH/graph.json","fetch_events":"https://pith.science/api/pith-number/72JWBN34NGDST6NS4X4YRLCVZH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/72JWBN34NGDST6NS4X4YRLCVZH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/72JWBN34NGDST6NS4X4YRLCVZH/action/storage_attestation","attest_author":"https://pith.science/pith/72JWBN34NGDST6NS4X4YRLCVZH/action/author_attestation","sign_citation":"https://pith.science/pith/72JWBN34NGDST6NS4X4YRLCVZH/action/citation_signature","submit_replication":"https://pith.science/pith/72JWBN34NGDST6NS4X4YRLCVZH/action/replication_record"}},"created_at":"2026-07-05T12:05:59.168955+00:00","updated_at":"2026-07-05T12:05:59.168955+00:00"}