{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:7W5KWKU2VGQVKAO3IE5Q5ZY4SC","short_pith_number":"pith:7W5KWKU2","schema_version":"1.0","canonical_sha256":"fdbaab2a9aa9a15501db413b0ee71c90ac6e964a1ed526425d30b1ea5cb1a23d","source":{"kind":"arxiv","id":"2508.10552","version":1},"attestation_state":"computed","paper":{"title":"When Language Overrules: Revealing Text Dominance in Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Haiyun Jiang, Huyu Wu, Meng Tang, Xinhan Zheng","submitted_at":"2025-08-14T11:44:52Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have demonstrated remarkable capabilities across a diverse range of multimodal tasks. However, these models suffer from a core problem known as text dominance: they depend heavily on text for their inference, while underutilizing other modalities. While prior work has acknowledged this phenomenon in vision-language tasks, often attributing it to data biases or model architectures. In this paper, we conduct the first systematic investigation of text dominance across diverse data modalities, including images, videos, audio, time-series, and graphs. To mea"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.10552","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-08-14T11:44:52Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"52fb752c4cb3f56f21cfa83f5345ce4bb8b05ef3f460053a12c4217c7563fb27","abstract_canon_sha256":"a3758f7458822f2740186c6fc3f8752c298218980d32ceaf177d212d6c2f8201"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:53:58.149387Z","signature_b64":"Sxacnhgpgor/5r5p8QckCiXF3A30/22dnJttO0YHWPzir8dEuxfXvyE9q3Huyi2RkkBpgkhfWHfSGzRRCaW7DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fdbaab2a9aa9a15501db413b0ee71c90ac6e964a1ed526425d30b1ea5cb1a23d","last_reissued_at":"2026-07-05T11:53:58.148964Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:53:58.148964Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"When Language Overrules: Revealing Text Dominance in Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Haiyun Jiang, Huyu Wu, Meng Tang, Xinhan Zheng","submitted_at":"2025-08-14T11:44:52Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have demonstrated remarkable capabilities across a diverse range of multimodal tasks. However, these models suffer from a core problem known as text dominance: they depend heavily on text for their inference, while underutilizing other modalities. While prior work has acknowledged this phenomenon in vision-language tasks, often attributing it to data biases or model architectures. In this paper, we conduct the first systematic investigation of text dominance across diverse data modalities, including images, videos, audio, time-series, and graphs. To mea"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.10552","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.10552/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.10552","created_at":"2026-07-05T11:53:58.149022+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.10552v1","created_at":"2026-07-05T11:53:58.149022+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.10552","created_at":"2026-07-05T11:53:58.149022+00:00"},{"alias_kind":"pith_short_12","alias_value":"7W5KWKU2VGQV","created_at":"2026-07-05T11:53:58.149022+00:00"},{"alias_kind":"pith_short_16","alias_value":"7W5KWKU2VGQVKAO3","created_at":"2026-07-05T11:53:58.149022+00:00"},{"alias_kind":"pith_short_8","alias_value":"7W5KWKU2","created_at":"2026-07-05T11:53:58.149022+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23052","citing_title":"CAAD: Contrastive Audio-Aware Distillation for Efficient Speech Language Models","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18924","citing_title":"Who Wins the Conflict? Mechanistic Interpretability of Text Bias in Audio LLMs","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10368","citing_title":"Speech Meets ELF: Audio Conditional Continuous-Target Diffusion for Speech Recognition and Translation","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02492","citing_title":"Token-Efficient Multimodal Reasoning via Image Prompt Packaging","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21326","citing_title":"MiMIC: Mitigating Visual Modality Collapse in Universal Multimodal Retrieval While Avoiding Semantic Misalignment","ref_index":102,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10039","citing_title":"Counting to Four is still a Chore for VLMs","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05117","citing_title":"Watch Before You Answer: Learning from Visually Grounded Post-Training","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16264","citing_title":"Information Router for Mitigating Modality Dominance in Vision-Language Models","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7W5KWKU2VGQVKAO3IE5Q5ZY4SC","json":"https://pith.science/pith/7W5KWKU2VGQVKAO3IE5Q5ZY4SC.json","graph_json":"https://pith.science/api/pith-number/7W5KWKU2VGQVKAO3IE5Q5ZY4SC/graph.json","events_json":"https://pith.science/api/pith-number/7W5KWKU2VGQVKAO3IE5Q5ZY4SC/events.json","paper":"https://pith.science/paper/7W5KWKU2"},"agent_actions":{"view_html":"https://pith.science/pith/7W5KWKU2VGQVKAO3IE5Q5ZY4SC","download_json":"https://pith.science/pith/7W5KWKU2VGQVKAO3IE5Q5ZY4SC.json","view_paper":"https://pith.science/paper/7W5KWKU2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.10552&json=true","fetch_graph":"https://pith.science/api/pith-number/7W5KWKU2VGQVKAO3IE5Q5ZY4SC/graph.json","fetch_events":"https://pith.science/api/pith-number/7W5KWKU2VGQVKAO3IE5Q5ZY4SC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7W5KWKU2VGQVKAO3IE5Q5ZY4SC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7W5KWKU2VGQVKAO3IE5Q5ZY4SC/action/storage_attestation","attest_author":"https://pith.science/pith/7W5KWKU2VGQVKAO3IE5Q5ZY4SC/action/author_attestation","sign_citation":"https://pith.science/pith/7W5KWKU2VGQVKAO3IE5Q5ZY4SC/action/citation_signature","submit_replication":"https://pith.science/pith/7W5KWKU2VGQVKAO3IE5Q5ZY4SC/action/replication_record"}},"created_at":"2026-07-05T11:53:58.149022+00:00","updated_at":"2026-07-05T11:53:58.149022+00:00"}