{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XCWBSSIAKIEBGQMEOKXPIWJFSH","short_pith_number":"pith:XCWBSSIA","schema_version":"1.0","canonical_sha256":"b8ac194900520813418472aef4592591fbe58c2490e1fa05095bdc0a8de1f772","source":{"kind":"arxiv","id":"2411.09273","version":1},"attestation_state":"computed","paper":{"title":"Cross-Modal Consistency in Multimodal Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bradley Hauer, Grzegorz Kondrak, Laks V.S. Lakshmanan, Muhammad Abdul-Mageed, Ning Shi, Senyu Li, Xiang Zhang, Zijun Wu","submitted_at":"2024-11-14T08:22:42Z","abstract_excerpt":"Recent developments in multimodal methodologies have marked the beginning of an exciting era for models adept at processing diverse data types, encompassing text, audio, and visual content. Models like GPT-4V, which merge computer vision with advanced language processing, exhibit extraordinary proficiency in handling intricate tasks that require a simultaneous understanding of both textual and visual information. Prior research efforts have meticulously evaluated the efficacy of these Vision Large Language Models (VLLMs) in various domains, including object detection, image captioning, and oth"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.09273","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-11-14T08:22:42Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"96ca183b9fbd284d515981882d44ed943281788507ae5d99c7213e9512fe97c3","abstract_canon_sha256":"c25f50fb805e5c152d9c721e401413906e4ba2e1bdf0a948401f4904e6175bb0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:35:23.740390Z","signature_b64":"3G1ijOj1fhtYxvpLWYupmsS4swTTdAf+dmWOnw/iLXOCMamyOSJe4NgzbWO0TyVvJ/gNQczHISSmBMpMwUkkBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b8ac194900520813418472aef4592591fbe58c2490e1fa05095bdc0a8de1f772","last_reissued_at":"2026-07-05T09:35:23.739893Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:35:23.739893Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Cross-Modal Consistency in Multimodal Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bradley Hauer, Grzegorz Kondrak, Laks V.S. Lakshmanan, Muhammad Abdul-Mageed, Ning Shi, Senyu Li, Xiang Zhang, Zijun Wu","submitted_at":"2024-11-14T08:22:42Z","abstract_excerpt":"Recent developments in multimodal methodologies have marked the beginning of an exciting era for models adept at processing diverse data types, encompassing text, audio, and visual content. Models like GPT-4V, which merge computer vision with advanced language processing, exhibit extraordinary proficiency in handling intricate tasks that require a simultaneous understanding of both textual and visual information. Prior research efforts have meticulously evaluated the efficacy of these Vision Large Language Models (VLLMs) in various domains, including object detection, image captioning, and oth"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.09273","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.09273/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.09273","created_at":"2026-07-05T09:35:23.739957+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.09273v1","created_at":"2026-07-05T09:35:23.739957+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.09273","created_at":"2026-07-05T09:35:23.739957+00:00"},{"alias_kind":"pith_short_12","alias_value":"XCWBSSIAKIEB","created_at":"2026-07-05T09:35:23.739957+00:00"},{"alias_kind":"pith_short_16","alias_value":"XCWBSSIAKIEBGQME","created_at":"2026-07-05T09:35:23.739957+00:00"},{"alias_kind":"pith_short_8","alias_value":"XCWBSSIA","created_at":"2026-07-05T09:35:23.739957+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11614","citing_title":"Information-Theoretic Decomposition for Multimodal Interaction Learning","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2510.15148","citing_title":"XModBench: Benchmarking Cross-Modal Capabilities and Consistency in Omni-Language Models","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XCWBSSIAKIEBGQMEOKXPIWJFSH","json":"https://pith.science/pith/XCWBSSIAKIEBGQMEOKXPIWJFSH.json","graph_json":"https://pith.science/api/pith-number/XCWBSSIAKIEBGQMEOKXPIWJFSH/graph.json","events_json":"https://pith.science/api/pith-number/XCWBSSIAKIEBGQMEOKXPIWJFSH/events.json","paper":"https://pith.science/paper/XCWBSSIA"},"agent_actions":{"view_html":"https://pith.science/pith/XCWBSSIAKIEBGQMEOKXPIWJFSH","download_json":"https://pith.science/pith/XCWBSSIAKIEBGQMEOKXPIWJFSH.json","view_paper":"https://pith.science/paper/XCWBSSIA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.09273&json=true","fetch_graph":"https://pith.science/api/pith-number/XCWBSSIAKIEBGQMEOKXPIWJFSH/graph.json","fetch_events":"https://pith.science/api/pith-number/XCWBSSIAKIEBGQMEOKXPIWJFSH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XCWBSSIAKIEBGQMEOKXPIWJFSH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XCWBSSIAKIEBGQMEOKXPIWJFSH/action/storage_attestation","attest_author":"https://pith.science/pith/XCWBSSIAKIEBGQMEOKXPIWJFSH/action/author_attestation","sign_citation":"https://pith.science/pith/XCWBSSIAKIEBGQMEOKXPIWJFSH/action/citation_signature","submit_replication":"https://pith.science/pith/XCWBSSIAKIEBGQMEOKXPIWJFSH/action/replication_record"}},"created_at":"2026-07-05T09:35:23.739957+00:00","updated_at":"2026-07-05T09:35:23.739957+00:00"}