{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:EWYL2RGGO57NTOM3CW4EGNXY6S","short_pith_number":"pith:EWYL2RGG","schema_version":"1.0","canonical_sha256":"25b0bd44c6777ed9b99b15b84336f8f4bc7f972e29c6865f6736dcd25a090514","source":{"kind":"arxiv","id":"2407.04482","version":2},"attestation_state":"computed","paper":{"title":"Controlling Whisper: Universal Acoustic Adversarial Attacks to Control Speech Foundation Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","eess.AS"],"primary_cat":"cs.SD","authors_text":"Mark Gales, Vyas Raina","submitted_at":"2024-07-05T13:04:31Z","abstract_excerpt":"Speech enabled foundation models, either in the form of flexible speech recognition based systems or audio-prompted large language models (LLMs), are becoming increasingly popular. One of the interesting aspects of these models is their ability to perform tasks other than automatic speech recognition (ASR) using an appropriate prompt. For example, the OpenAI Whisper model can perform both speech transcription and speech translation. With the development of audio-prompted LLMs there is the potential for even greater control options. In this work we demonstrate that with this greater flexibility"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.04482","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SD","submitted_at":"2024-07-05T13:04:31Z","cross_cats_sorted":["cs.CL","eess.AS"],"title_canon_sha256":"86d3f06a5728e04a785368a6aad80790540e56064265f16cb7e7b570b426f2ac","abstract_canon_sha256":"1d8f59121f8ca80bed532d564d59b52a5bdb44459ef5f248b8b71525829d5af1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:19:11.118658Z","signature_b64":"HK0AELIJE4Q/gfuhdC+WrQrIRTxyHXBSLGDsppxQ4Ou9UKCznGKeg6ZjvNQ0Oi2CSpU+7dIMGvzbSdmrm+o7AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"25b0bd44c6777ed9b99b15b84336f8f4bc7f972e29c6865f6736dcd25a090514","last_reissued_at":"2026-07-05T09:19:11.118184Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:19:11.118184Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Controlling Whisper: Universal Acoustic Adversarial Attacks to Control Speech Foundation Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","eess.AS"],"primary_cat":"cs.SD","authors_text":"Mark Gales, Vyas Raina","submitted_at":"2024-07-05T13:04:31Z","abstract_excerpt":"Speech enabled foundation models, either in the form of flexible speech recognition based systems or audio-prompted large language models (LLMs), are becoming increasingly popular. One of the interesting aspects of these models is their ability to perform tasks other than automatic speech recognition (ASR) using an appropriate prompt. For example, the OpenAI Whisper model can perform both speech transcription and speech translation. With the development of audio-prompted LLMs there is the potential for even greater control options. In this work we demonstrate that with this greater flexibility"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.04482","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.04482/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.04482","created_at":"2026-07-05T09:19:11.118243+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.04482v2","created_at":"2026-07-05T09:19:11.118243+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.04482","created_at":"2026-07-05T09:19:11.118243+00:00"},{"alias_kind":"pith_short_12","alias_value":"EWYL2RGGO57N","created_at":"2026-07-05T09:19:11.118243+00:00"},{"alias_kind":"pith_short_16","alias_value":"EWYL2RGGO57NTOM3","created_at":"2026-07-05T09:19:11.118243+00:00"},{"alias_kind":"pith_short_8","alias_value":"EWYL2RGG","created_at":"2026-07-05T09:19:11.118243+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.00548","citing_title":"Con Instruction: Universal Jailbreaking of Multimodal Large Language Models via Non-Textual Modalities","ref_index":25,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EWYL2RGGO57NTOM3CW4EGNXY6S","json":"https://pith.science/pith/EWYL2RGGO57NTOM3CW4EGNXY6S.json","graph_json":"https://pith.science/api/pith-number/EWYL2RGGO57NTOM3CW4EGNXY6S/graph.json","events_json":"https://pith.science/api/pith-number/EWYL2RGGO57NTOM3CW4EGNXY6S/events.json","paper":"https://pith.science/paper/EWYL2RGG"},"agent_actions":{"view_html":"https://pith.science/pith/EWYL2RGGO57NTOM3CW4EGNXY6S","download_json":"https://pith.science/pith/EWYL2RGGO57NTOM3CW4EGNXY6S.json","view_paper":"https://pith.science/paper/EWYL2RGG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.04482&json=true","fetch_graph":"https://pith.science/api/pith-number/EWYL2RGGO57NTOM3CW4EGNXY6S/graph.json","fetch_events":"https://pith.science/api/pith-number/EWYL2RGGO57NTOM3CW4EGNXY6S/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EWYL2RGGO57NTOM3CW4EGNXY6S/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EWYL2RGGO57NTOM3CW4EGNXY6S/action/storage_attestation","attest_author":"https://pith.science/pith/EWYL2RGGO57NTOM3CW4EGNXY6S/action/author_attestation","sign_citation":"https://pith.science/pith/EWYL2RGGO57NTOM3CW4EGNXY6S/action/citation_signature","submit_replication":"https://pith.science/pith/EWYL2RGGO57NTOM3CW4EGNXY6S/action/replication_record"}},"created_at":"2026-07-05T09:19:11.118243+00:00","updated_at":"2026-07-05T09:19:11.118243+00:00"}