{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:736BMABIOB454AMOPXAQ2N2M64","short_pith_number":"pith:736BMABI","schema_version":"1.0","canonical_sha256":"fefc1600287079de018e7dc10d374cf703c7e4796e59edfb9e82f54746e295d7","source":{"kind":"arxiv","id":"2411.18620","version":2},"attestation_state":"computed","paper":{"title":"Cross-modal Information Flow in Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CV"],"primary_cat":"cs.AI","authors_text":"Ekaterina Shutova, Fengze Han, Srishti Yadav, Zhi Zhang","submitted_at":"2024-11-27T18:59:26Z","abstract_excerpt":"The recent advancements in auto-regressive multimodal large language models (MLLMs) have demonstrated promising progress for vision-language tasks. While there exists a variety of studies investigating the processing of linguistic information within large language models, little is currently known about the inner working mechanism of MLLMs and how linguistic and visual information interact within these models. In this study, we aim to fill this gap by examining the information flow between different modalities -- language and vision -- in MLLMs, focusing on visual question answering. Specifica"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.18620","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-11-27T18:59:26Z","cross_cats_sorted":["cs.CL","cs.CV"],"title_canon_sha256":"64fc462d6f6cb359dc6d72ccc7f6c18a6428a437fc672ffb52e02b0e209c4b13","abstract_canon_sha256":"857aba631cfc19f6b7d4a5b70cbd65070512d76575c4f26ad5e81be805341253"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:39:46.685434Z","signature_b64":"rsDsohj2+7/6/cLiPLJTNmEDgQBLFx8pM31WbIuccoVtIwIGcq33XbleMdZJfMTsgD7kfaB+lpQMrF+KEXEiAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fefc1600287079de018e7dc10d374cf703c7e4796e59edfb9e82f54746e295d7","last_reissued_at":"2026-07-05T10:39:46.684932Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:39:46.684932Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Cross-modal Information Flow in Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CV"],"primary_cat":"cs.AI","authors_text":"Ekaterina Shutova, Fengze Han, Srishti Yadav, Zhi Zhang","submitted_at":"2024-11-27T18:59:26Z","abstract_excerpt":"The recent advancements in auto-regressive multimodal large language models (MLLMs) have demonstrated promising progress for vision-language tasks. While there exists a variety of studies investigating the processing of linguistic information within large language models, little is currently known about the inner working mechanism of MLLMs and how linguistic and visual information interact within these models. In this study, we aim to fill this gap by examining the information flow between different modalities -- language and vision -- in MLLMs, focusing on visual question answering. Specifica"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.18620","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.18620/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.18620","created_at":"2026-07-05T10:39:46.685003+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.18620v2","created_at":"2026-07-05T10:39:46.685003+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.18620","created_at":"2026-07-05T10:39:46.685003+00:00"},{"alias_kind":"pith_short_12","alias_value":"736BMABIOB45","created_at":"2026-07-05T10:39:46.685003+00:00"},{"alias_kind":"pith_short_16","alias_value":"736BMABIOB454AMO","created_at":"2026-07-05T10:39:46.685003+00:00"},{"alias_kind":"pith_short_8","alias_value":"736BMABI","created_at":"2026-07-05T10:39:46.685003+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2503.14075","citing_title":"Growing a Multi-head Twig via Distillation and Reinforcement Learning to Accelerate Large Vision-Language Models","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19250","citing_title":"Causal Evidence for Attention Head Imbalance in Modality Conflict Hallucination","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/736BMABIOB454AMOPXAQ2N2M64","json":"https://pith.science/pith/736BMABIOB454AMOPXAQ2N2M64.json","graph_json":"https://pith.science/api/pith-number/736BMABIOB454AMOPXAQ2N2M64/graph.json","events_json":"https://pith.science/api/pith-number/736BMABIOB454AMOPXAQ2N2M64/events.json","paper":"https://pith.science/paper/736BMABI"},"agent_actions":{"view_html":"https://pith.science/pith/736BMABIOB454AMOPXAQ2N2M64","download_json":"https://pith.science/pith/736BMABIOB454AMOPXAQ2N2M64.json","view_paper":"https://pith.science/paper/736BMABI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.18620&json=true","fetch_graph":"https://pith.science/api/pith-number/736BMABIOB454AMOPXAQ2N2M64/graph.json","fetch_events":"https://pith.science/api/pith-number/736BMABIOB454AMOPXAQ2N2M64/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/736BMABIOB454AMOPXAQ2N2M64/action/timestamp_anchor","attest_storage":"https://pith.science/pith/736BMABIOB454AMOPXAQ2N2M64/action/storage_attestation","attest_author":"https://pith.science/pith/736BMABIOB454AMOPXAQ2N2M64/action/author_attestation","sign_citation":"https://pith.science/pith/736BMABIOB454AMOPXAQ2N2M64/action/citation_signature","submit_replication":"https://pith.science/pith/736BMABIOB454AMOPXAQ2N2M64/action/replication_record"}},"created_at":"2026-07-05T10:39:46.685003+00:00","updated_at":"2026-07-05T10:39:46.685003+00:00"}