{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:SQ5WN7RZEZLYSOMQMLY4EZHBZC","short_pith_number":"pith:SQ5WN7RZ","schema_version":"1.0","canonical_sha256":"943b66fe39265789399062f1c264e1c89e71c34d0072276104f1f5adc69faa8b","source":{"kind":"arxiv","id":"2507.06256","version":1},"attestation_state":"computed","paper":{"title":"Attacker's Noise Can Manipulate Your Audio-based LLM in the Real World","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.SD","eess.AS"],"primary_cat":"cs.CR","authors_text":"Lun Wang, Rajiv Mathews, Soheil Feizi, Vinu Sankar Sadasivan","submitted_at":"2025-07-07T07:29:52Z","abstract_excerpt":"This paper investigates the real-world vulnerabilities of audio-based large language models (ALLMs), such as Qwen2-Audio. We first demonstrate that an adversary can craft stealthy audio perturbations to manipulate ALLMs into exhibiting specific targeted behaviors, such as eliciting responses to wake-keywords (e.g., \"Hey Qwen\"), or triggering harmful behaviors (e.g. \"Change my calendar event\"). Subsequently, we show that playing adversarial background noise during user interaction with the ALLMs can significantly degrade the response quality. Crucially, our research illustrates the scalability "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.06256","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2025-07-07T07:29:52Z","cross_cats_sorted":["cs.AI","cs.SD","eess.AS"],"title_canon_sha256":"97bc0bc337dcc0c9157a745b150dad2feca28dcf12c76a3c1d7598465ce5874c","abstract_canon_sha256":"3876fcf68817067793ed993f98ca49ec5c485e3ce49bcc1346aba7527c4bd719"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:34:12.644589Z","signature_b64":"r09AX/NAuhn19Sp2ubZ5qmBUB0L3fOA5qnkGXWxCn27wmxwSy5VvIyypXfi3/VBFAaCG70T+oIXhMSDO3AXMBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"943b66fe39265789399062f1c264e1c89e71c34d0072276104f1f5adc69faa8b","last_reissued_at":"2026-07-05T11:34:12.644069Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:34:12.644069Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Attacker's Noise Can Manipulate Your Audio-based LLM in the Real World","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.SD","eess.AS"],"primary_cat":"cs.CR","authors_text":"Lun Wang, Rajiv Mathews, Soheil Feizi, Vinu Sankar Sadasivan","submitted_at":"2025-07-07T07:29:52Z","abstract_excerpt":"This paper investigates the real-world vulnerabilities of audio-based large language models (ALLMs), such as Qwen2-Audio. We first demonstrate that an adversary can craft stealthy audio perturbations to manipulate ALLMs into exhibiting specific targeted behaviors, such as eliciting responses to wake-keywords (e.g., \"Hey Qwen\"), or triggering harmful behaviors (e.g. \"Change my calendar event\"). Subsequently, we show that playing adversarial background noise during user interaction with the ALLMs can significantly degrade the response quality. Crucially, our research illustrates the scalability "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.06256","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.06256/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.06256","created_at":"2026-07-05T11:34:12.644129+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.06256v1","created_at":"2026-07-05T11:34:12.644129+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.06256","created_at":"2026-07-05T11:34:12.644129+00:00"},{"alias_kind":"pith_short_12","alias_value":"SQ5WN7RZEZLY","created_at":"2026-07-05T11:34:12.644129+00:00"},{"alias_kind":"pith_short_16","alias_value":"SQ5WN7RZEZLYSOMQ","created_at":"2026-07-05T11:34:12.644129+00:00"},{"alias_kind":"pith_short_8","alias_value":"SQ5WN7RZ","created_at":"2026-07-05T11:34:12.644129+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11219","citing_title":"Afrispeech Semantics: Evaluating Audio Semantic Reasoning in Spoken Language Models Across Domains and Accents","ref_index":223,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20519","citing_title":"Codec-Robust Attacks on Audio LLMs","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20266","citing_title":"A Survey of Large Audio Language Models: Generalization, Trustworthiness, and Outlook","ref_index":149,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20519","citing_title":"Codec-Robust Attacks on Audio LLMs","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04700","citing_title":"Sparse Tokens Suffice: Jailbreaking Audio Language Models via Token-Aware Gradient Optimization","ref_index":49,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SQ5WN7RZEZLYSOMQMLY4EZHBZC","json":"https://pith.science/pith/SQ5WN7RZEZLYSOMQMLY4EZHBZC.json","graph_json":"https://pith.science/api/pith-number/SQ5WN7RZEZLYSOMQMLY4EZHBZC/graph.json","events_json":"https://pith.science/api/pith-number/SQ5WN7RZEZLYSOMQMLY4EZHBZC/events.json","paper":"https://pith.science/paper/SQ5WN7RZ"},"agent_actions":{"view_html":"https://pith.science/pith/SQ5WN7RZEZLYSOMQMLY4EZHBZC","download_json":"https://pith.science/pith/SQ5WN7RZEZLYSOMQMLY4EZHBZC.json","view_paper":"https://pith.science/paper/SQ5WN7RZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.06256&json=true","fetch_graph":"https://pith.science/api/pith-number/SQ5WN7RZEZLYSOMQMLY4EZHBZC/graph.json","fetch_events":"https://pith.science/api/pith-number/SQ5WN7RZEZLYSOMQMLY4EZHBZC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SQ5WN7RZEZLYSOMQMLY4EZHBZC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SQ5WN7RZEZLYSOMQMLY4EZHBZC/action/storage_attestation","attest_author":"https://pith.science/pith/SQ5WN7RZEZLYSOMQMLY4EZHBZC/action/author_attestation","sign_citation":"https://pith.science/pith/SQ5WN7RZEZLYSOMQMLY4EZHBZC/action/citation_signature","submit_replication":"https://pith.science/pith/SQ5WN7RZEZLYSOMQMLY4EZHBZC/action/replication_record"}},"created_at":"2026-07-05T11:34:12.644129+00:00","updated_at":"2026-07-05T11:34:12.644129+00:00"}