{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ZIZE3USXKEAPDJICNF23TBAWS4","short_pith_number":"pith:ZIZE3USX","schema_version":"1.0","canonical_sha256":"ca324dd2575100f1a5026975b98416972a05ec046e3c9f1629e0206f3789af9c","source":{"kind":"arxiv","id":"2410.23861","version":1},"attestation_state":"computed","paper":{"title":"Audio Is the Achilles' Heel: Red Teaming Audio Large Multimodal Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MM","cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Ehsan Shareghi, Gholamreza Haffari, Hao Yang, Lizhen Qu","submitted_at":"2024-10-31T12:11:17Z","abstract_excerpt":"Large Multimodal Models (LMMs) have demonstrated the ability to interact with humans under real-world conditions by combining Large Language Models (LLMs) and modality encoders to align multimodal information (visual and auditory) with text. However, such models raise new safety challenges of whether models that are safety-aligned on text also exhibit consistent safeguards for multimodal inputs. Despite recent safety-alignment research on vision LMMs, the safety of audio LMMs remains under-explored. In this work, we comprehensively red team the safety of five advanced audio LMMs under three se"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.23861","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-10-31T12:11:17Z","cross_cats_sorted":["cs.MM","cs.SD","eess.AS"],"title_canon_sha256":"fe5a9d23413f35ea11cc1cb54efd4fbbdc4fdb513017317a9cba768251967f18","abstract_canon_sha256":"96445aae79f682316d7b28c608948e1b8c57fdca4fafa61ab9844bf34ade6e33"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:29:06.231537Z","signature_b64":"f+tQChvblh6Wa5VRyVL+QBhAyoLB32bW4lEmcsJQidSGtXFXPXYy85x+hih888Eo5sf7EI/HFXRHCBJtzKppAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ca324dd2575100f1a5026975b98416972a05ec046e3c9f1629e0206f3789af9c","last_reissued_at":"2026-07-05T09:29:06.231035Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:29:06.231035Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Audio Is the Achilles' Heel: Red Teaming Audio Large Multimodal Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MM","cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Ehsan Shareghi, Gholamreza Haffari, Hao Yang, Lizhen Qu","submitted_at":"2024-10-31T12:11:17Z","abstract_excerpt":"Large Multimodal Models (LMMs) have demonstrated the ability to interact with humans under real-world conditions by combining Large Language Models (LLMs) and modality encoders to align multimodal information (visual and auditory) with text. However, such models raise new safety challenges of whether models that are safety-aligned on text also exhibit consistent safeguards for multimodal inputs. Despite recent safety-alignment research on vision LMMs, the safety of audio LMMs remains under-explored. In this work, we comprehensively red team the safety of five advanced audio LMMs under three se"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.23861","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.23861/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.23861","created_at":"2026-07-05T09:29:06.231094+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.23861v1","created_at":"2026-07-05T09:29:06.231094+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.23861","created_at":"2026-07-05T09:29:06.231094+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZIZE3USXKEAP","created_at":"2026-07-05T09:29:06.231094+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZIZE3USXKEAPDJIC","created_at":"2026-07-05T09:29:06.231094+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZIZE3USX","created_at":"2026-07-05T09:29:06.231094+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.11968","citing_title":"Watch, Listen, Understand, Mislead: Tri-modal Adversarial Attacks on Short Videos for Content Appropriateness Evaluation","ref_index":41,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZIZE3USXKEAPDJICNF23TBAWS4","json":"https://pith.science/pith/ZIZE3USXKEAPDJICNF23TBAWS4.json","graph_json":"https://pith.science/api/pith-number/ZIZE3USXKEAPDJICNF23TBAWS4/graph.json","events_json":"https://pith.science/api/pith-number/ZIZE3USXKEAPDJICNF23TBAWS4/events.json","paper":"https://pith.science/paper/ZIZE3USX"},"agent_actions":{"view_html":"https://pith.science/pith/ZIZE3USXKEAPDJICNF23TBAWS4","download_json":"https://pith.science/pith/ZIZE3USXKEAPDJICNF23TBAWS4.json","view_paper":"https://pith.science/paper/ZIZE3USX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.23861&json=true","fetch_graph":"https://pith.science/api/pith-number/ZIZE3USXKEAPDJICNF23TBAWS4/graph.json","fetch_events":"https://pith.science/api/pith-number/ZIZE3USXKEAPDJICNF23TBAWS4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZIZE3USXKEAPDJICNF23TBAWS4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZIZE3USXKEAPDJICNF23TBAWS4/action/storage_attestation","attest_author":"https://pith.science/pith/ZIZE3USXKEAPDJICNF23TBAWS4/action/author_attestation","sign_citation":"https://pith.science/pith/ZIZE3USXKEAPDJICNF23TBAWS4/action/citation_signature","submit_replication":"https://pith.science/pith/ZIZE3USXKEAPDJICNF23TBAWS4/action/replication_record"}},"created_at":"2026-07-05T09:29:06.231094+00:00","updated_at":"2026-07-05T09:29:06.231094+00:00"}