{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:EUPBRWCQBFBM7KVA6EZF7663DN","short_pith_number":"pith:EUPBRWCQ","schema_version":"1.0","canonical_sha256":"251e18d8500942cfaaa0f1325ffbdb1b5e19b6846f5d1ab249036b4404503bb9","source":{"kind":"arxiv","id":"2305.04160","version":3},"attestation_state":"computed","paper":{"title":"X-LLM: Bootstrapping Advanced Large Language Models by Treating Multi-Modalities as Foreign Languages","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CV","eess.AS"],"primary_cat":"cs.CL","authors_text":"Bo Xu, Feilong Chen, Haozhi Zhao, Jing Shi, Minglun Han, Qingyang Zhang, Shuang Xu","submitted_at":"2023-05-07T02:25:42Z","abstract_excerpt":"Large language models (LLMs) have demonstrated remarkable language abilities. GPT-4, based on advanced LLMs, exhibits extraordinary multimodal capabilities beyond previous visual language models. We attribute this to the use of more advanced LLMs compared with previous multimodal models. Unfortunately, the model architecture and training strategies of GPT-4 are unknown. To endow LLMs with multimodal capabilities, we propose X-LLM, which converts Multi-modalities (images, speech, videos) into foreign languages using X2L interfaces and inputs them into a large Language model (ChatGLM). Specifica"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.04160","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2023-05-07T02:25:42Z","cross_cats_sorted":["cs.AI","cs.CV","eess.AS"],"title_canon_sha256":"e587d00dbe443d84c80acc02be3cacd64773008540c7a9c325d74595214d4336","abstract_canon_sha256":"6e9e413e7ff8a04d29714d2fb11485fa9e151ec57c6b3b991cc62ad20a2f49e3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:12:02.501037Z","signature_b64":"vHj+qgjLD7iZ1bq0W5SJ8utiV/jCJi2IQs5hzQC+6JfVvQEK/iYyZunj1pQcQVNhcXLR7PdnxJ0KhhdapLTgCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"251e18d8500942cfaaa0f1325ffbdb1b5e19b6846f5d1ab249036b4404503bb9","last_reissued_at":"2026-07-05T06:12:02.500696Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:12:02.500696Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"X-LLM: Bootstrapping Advanced Large Language Models by Treating Multi-Modalities as Foreign Languages","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CV","eess.AS"],"primary_cat":"cs.CL","authors_text":"Bo Xu, Feilong Chen, Haozhi Zhao, Jing Shi, Minglun Han, Qingyang Zhang, Shuang Xu","submitted_at":"2023-05-07T02:25:42Z","abstract_excerpt":"Large language models (LLMs) have demonstrated remarkable language abilities. GPT-4, based on advanced LLMs, exhibits extraordinary multimodal capabilities beyond previous visual language models. We attribute this to the use of more advanced LLMs compared with previous multimodal models. Unfortunately, the model architecture and training strategies of GPT-4 are unknown. To endow LLMs with multimodal capabilities, we propose X-LLM, which converts Multi-modalities (images, speech, videos) into foreign languages using X2L interfaces and inputs them into a large Language model (ChatGLM). Specifica"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.04160","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.04160/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.04160","created_at":"2026-07-05T06:12:02.500752+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.04160v3","created_at":"2026-07-05T06:12:02.500752+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.04160","created_at":"2026-07-05T06:12:02.500752+00:00"},{"alias_kind":"pith_short_12","alias_value":"EUPBRWCQBFBM","created_at":"2026-07-05T06:12:02.500752+00:00"},{"alias_kind":"pith_short_16","alias_value":"EUPBRWCQBFBM7KVA","created_at":"2026-07-05T06:12:02.500752+00:00"},{"alias_kind":"pith_short_8","alias_value":"EUPBRWCQ","created_at":"2026-07-05T06:12:02.500752+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22352","citing_title":"On the Sparsity-Storage-Accuracy Tradeoff in Parsimoniously Activated Dictionary Learning","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2504.08528","citing_title":"On The Landscape of Spoken Language Models: A Comprehensive Survey","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2603.25120","citing_title":"DFLOP: A Data-driven Framework for Multimodal LLM Training Pipeline Optimization","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2507.23511","citing_title":"MECAT: A Multi-Experts Constructed Benchmark for Fine-Grained Audio Understanding Tasks","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2310.13289","citing_title":"SALMONN: Towards Generic Hearing Abilities for Large Language Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2602.12286","citing_title":"Mind the Gap No More: Achieving Zero-Gap Multimodal Integration via One Tokenizer","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2306.13549","citing_title":"A Survey on Multimodal Large Language Models","ref_index":72,"is_internal_anchor":false},{"citing_arxiv_id":"2311.10122","citing_title":"Video-LLaVA: Learning United Visual Representation by Alignment Before Projection","ref_index":113,"is_internal_anchor":false},{"citing_arxiv_id":"2309.07864","citing_title":"The Rise and Potential of Large Language Model Based Agents: A Survey","ref_index":297,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14951","citing_title":"RaTA-Tool: Retrieval-based Tool Selection with Multimodal Large Language Models","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EUPBRWCQBFBM7KVA6EZF7663DN","json":"https://pith.science/pith/EUPBRWCQBFBM7KVA6EZF7663DN.json","graph_json":"https://pith.science/api/pith-number/EUPBRWCQBFBM7KVA6EZF7663DN/graph.json","events_json":"https://pith.science/api/pith-number/EUPBRWCQBFBM7KVA6EZF7663DN/events.json","paper":"https://pith.science/paper/EUPBRWCQ"},"agent_actions":{"view_html":"https://pith.science/pith/EUPBRWCQBFBM7KVA6EZF7663DN","download_json":"https://pith.science/pith/EUPBRWCQBFBM7KVA6EZF7663DN.json","view_paper":"https://pith.science/paper/EUPBRWCQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.04160&json=true","fetch_graph":"https://pith.science/api/pith-number/EUPBRWCQBFBM7KVA6EZF7663DN/graph.json","fetch_events":"https://pith.science/api/pith-number/EUPBRWCQBFBM7KVA6EZF7663DN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EUPBRWCQBFBM7KVA6EZF7663DN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EUPBRWCQBFBM7KVA6EZF7663DN/action/storage_attestation","attest_author":"https://pith.science/pith/EUPBRWCQBFBM7KVA6EZF7663DN/action/author_attestation","sign_citation":"https://pith.science/pith/EUPBRWCQBFBM7KVA6EZF7663DN/action/citation_signature","submit_replication":"https://pith.science/pith/EUPBRWCQBFBM7KVA6EZF7663DN/action/replication_record"}},"created_at":"2026-07-05T06:12:02.500752+00:00","updated_at":"2026-07-05T06:12:02.500752+00:00"}