{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:3A6OVADAULWES6SYGZEZFA3UDG","short_pith_number":"pith:3A6OVADA","schema_version":"1.0","canonical_sha256":"d83cea8060a2ec497a5836499283741986c46817e94be8bef1c29492d1618632","source":{"kind":"arxiv","id":"2311.09363","version":2},"attestation_state":"computed","paper":{"title":"Investigating the Emergent Audio Classification Ability of ASR Foundation Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Adian Liusie, Kate M. Knill, Mark J. F. Gales, Rao Ma","submitted_at":"2023-11-15T20:52:56Z","abstract_excerpt":"Text and vision foundation models can perform many tasks in a zero-shot setting, a desirable property that enables these systems to be applied in general and low-resource settings. There has been far less work, however, on the zero-shot abilities of ASR foundation models, with these systems typically fine-tuned to specific tasks or constrained to applications that match their training criterion and data annotation. In this work we investigate the ability of Whisper and MMS, ASR foundation models trained primarily for speech recognition, to perform zero-shot audio classification. We use simple "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.09363","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-11-15T20:52:56Z","cross_cats_sorted":[],"title_canon_sha256":"f505a59680ba2fcff035cd5d99e0fb42059c391ebf419f196fbf2f84fbf64e86","abstract_canon_sha256":"d15ee1a1430371a5b7e6b7d9168a8e168eb627fa4c89b274ae4db2fd67d38014"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:01:34.241822Z","signature_b64":"fisNP1WKDopfb+Nd6FWHA9R++UA1dsCybp7HabI5t5QM54/dALNyTHZVelbL3crC8ATAfNJVPWQdgrma8XOdCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d83cea8060a2ec497a5836499283741986c46817e94be8bef1c29492d1618632","last_reissued_at":"2026-07-05T08:01:34.241336Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:01:34.241336Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Investigating the Emergent Audio Classification Ability of ASR Foundation Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Adian Liusie, Kate M. Knill, Mark J. F. Gales, Rao Ma","submitted_at":"2023-11-15T20:52:56Z","abstract_excerpt":"Text and vision foundation models can perform many tasks in a zero-shot setting, a desirable property that enables these systems to be applied in general and low-resource settings. There has been far less work, however, on the zero-shot abilities of ASR foundation models, with these systems typically fine-tuned to specific tasks or constrained to applications that match their training criterion and data annotation. In this work we investigate the ability of Whisper and MMS, ASR foundation models trained primarily for speech recognition, to perform zero-shot audio classification. We use simple "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.09363","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.09363/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.09363","created_at":"2026-07-05T08:01:34.241403+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.09363v2","created_at":"2026-07-05T08:01:34.241403+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.09363","created_at":"2026-07-05T08:01:34.241403+00:00"},{"alias_kind":"pith_short_12","alias_value":"3A6OVADAULWE","created_at":"2026-07-05T08:01:34.241403+00:00"},{"alias_kind":"pith_short_16","alias_value":"3A6OVADAULWES6SY","created_at":"2026-07-05T08:01:34.241403+00:00"},{"alias_kind":"pith_short_8","alias_value":"3A6OVADA","created_at":"2026-07-05T08:01:34.241403+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.21809","citing_title":"Voice Quality Dimensions as Interpretable Primitives for Speaking Style for Atypical Speech and Affect","ref_index":44,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3A6OVADAULWES6SYGZEZFA3UDG","json":"https://pith.science/pith/3A6OVADAULWES6SYGZEZFA3UDG.json","graph_json":"https://pith.science/api/pith-number/3A6OVADAULWES6SYGZEZFA3UDG/graph.json","events_json":"https://pith.science/api/pith-number/3A6OVADAULWES6SYGZEZFA3UDG/events.json","paper":"https://pith.science/paper/3A6OVADA"},"agent_actions":{"view_html":"https://pith.science/pith/3A6OVADAULWES6SYGZEZFA3UDG","download_json":"https://pith.science/pith/3A6OVADAULWES6SYGZEZFA3UDG.json","view_paper":"https://pith.science/paper/3A6OVADA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.09363&json=true","fetch_graph":"https://pith.science/api/pith-number/3A6OVADAULWES6SYGZEZFA3UDG/graph.json","fetch_events":"https://pith.science/api/pith-number/3A6OVADAULWES6SYGZEZFA3UDG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3A6OVADAULWES6SYGZEZFA3UDG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3A6OVADAULWES6SYGZEZFA3UDG/action/storage_attestation","attest_author":"https://pith.science/pith/3A6OVADAULWES6SYGZEZFA3UDG/action/author_attestation","sign_citation":"https://pith.science/pith/3A6OVADAULWES6SYGZEZFA3UDG/action/citation_signature","submit_replication":"https://pith.science/pith/3A6OVADAULWES6SYGZEZFA3UDG/action/replication_record"}},"created_at":"2026-07-05T08:01:34.241403+00:00","updated_at":"2026-07-05T08:01:34.241403+00:00"}