{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MBN5BMEUGSEHG6DX4NR2DKTR7G","short_pith_number":"pith:MBN5BMEU","schema_version":"1.0","canonical_sha256":"605bd0b0943488737877e363a1aa71f9a49cc6156a4fa72f2ab5bf5477fb323a","source":{"kind":"arxiv","id":"2411.05056","version":1},"attestation_state":"computed","paper":{"title":"Seeing is Deceiving: Exploitation of Visual Pathways in Multi-Modal Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Arlo Octavia, Ave Giulietta, Linda Laurier, Meade Cleti, Pete Janowczyk","submitted_at":"2024-11-07T16:21:18Z","abstract_excerpt":"Multi-Modal Language Models (MLLMs) have transformed artificial intelligence by combining visual and text data, making applications like image captioning, visual question answering, and multi-modal content creation possible. This ability to understand and work with complex information has made MLLMs useful in areas such as healthcare, autonomous systems, and digital content. However, integrating multiple types of data also creates security risks. Attackers can manipulate either the visual or text inputs, or both, to make the model produce unintended or even harmful responses. This paper review"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.05056","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2024-11-07T16:21:18Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"8839892447ca3e4a66a4a619a4c0ad7b8bd173c4ff81db853d4bd41c2595bb50","abstract_canon_sha256":"04b6c14f209d1ba29fc3bd21cfacdf9bdb9613a94943e56f057bb890f2e3325c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:32:52.498901Z","signature_b64":"p3oYq2PeKQrfw8chJtxI+3nTxW9IXhKnYNK0t7chu1RVILujqclwi2kk7YK4XVeculjhn/c1lkOm/FOfrBEmDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"605bd0b0943488737877e363a1aa71f9a49cc6156a4fa72f2ab5bf5477fb323a","last_reissued_at":"2026-07-05T09:32:52.498361Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:32:52.498361Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Seeing is Deceiving: Exploitation of Visual Pathways in Multi-Modal Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Arlo Octavia, Ave Giulietta, Linda Laurier, Meade Cleti, Pete Janowczyk","submitted_at":"2024-11-07T16:21:18Z","abstract_excerpt":"Multi-Modal Language Models (MLLMs) have transformed artificial intelligence by combining visual and text data, making applications like image captioning, visual question answering, and multi-modal content creation possible. This ability to understand and work with complex information has made MLLMs useful in areas such as healthcare, autonomous systems, and digital content. However, integrating multiple types of data also creates security risks. Attackers can manipulate either the visual or text inputs, or both, to make the model produce unintended or even harmful responses. This paper review"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.05056","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.05056/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.05056","created_at":"2026-07-05T09:32:52.498418+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.05056v1","created_at":"2026-07-05T09:32:52.498418+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.05056","created_at":"2026-07-05T09:32:52.498418+00:00"},{"alias_kind":"pith_short_12","alias_value":"MBN5BMEUGSEH","created_at":"2026-07-05T09:32:52.498418+00:00"},{"alias_kind":"pith_short_16","alias_value":"MBN5BMEUGSEHG6DX","created_at":"2026-07-05T09:32:52.498418+00:00"},{"alias_kind":"pith_short_8","alias_value":"MBN5BMEU","created_at":"2026-07-05T09:32:52.498418+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.15435","citing_title":"ORCA: An Agentic Reasoning Framework for Hallucination and Adversarial Robustness in Vision-Language Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2509.15435","citing_title":"ORCA: An Agentic Reasoning Framework for Hallucination and Adversarial Robustness in Vision-Language Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2601.18842","citing_title":"GUIGuard-Bench: Toward a General Evaluation for Privacy-Preserving GUI Agents","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MBN5BMEUGSEHG6DX4NR2DKTR7G","json":"https://pith.science/pith/MBN5BMEUGSEHG6DX4NR2DKTR7G.json","graph_json":"https://pith.science/api/pith-number/MBN5BMEUGSEHG6DX4NR2DKTR7G/graph.json","events_json":"https://pith.science/api/pith-number/MBN5BMEUGSEHG6DX4NR2DKTR7G/events.json","paper":"https://pith.science/paper/MBN5BMEU"},"agent_actions":{"view_html":"https://pith.science/pith/MBN5BMEUGSEHG6DX4NR2DKTR7G","download_json":"https://pith.science/pith/MBN5BMEUGSEHG6DX4NR2DKTR7G.json","view_paper":"https://pith.science/paper/MBN5BMEU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.05056&json=true","fetch_graph":"https://pith.science/api/pith-number/MBN5BMEUGSEHG6DX4NR2DKTR7G/graph.json","fetch_events":"https://pith.science/api/pith-number/MBN5BMEUGSEHG6DX4NR2DKTR7G/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MBN5BMEUGSEHG6DX4NR2DKTR7G/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MBN5BMEUGSEHG6DX4NR2DKTR7G/action/storage_attestation","attest_author":"https://pith.science/pith/MBN5BMEUGSEHG6DX4NR2DKTR7G/action/author_attestation","sign_citation":"https://pith.science/pith/MBN5BMEUGSEHG6DX4NR2DKTR7G/action/citation_signature","submit_replication":"https://pith.science/pith/MBN5BMEUGSEHG6DX4NR2DKTR7G/action/replication_record"}},"created_at":"2026-07-05T09:32:52.498418+00:00","updated_at":"2026-07-05T09:32:52.498418+00:00"}