{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:Q7S5RESRIJDGSHTC3U7UVPWJTG","short_pith_number":"pith:Q7S5RESR","schema_version":"1.0","canonical_sha256":"87e5d892514246691e62dd3f4abec99992b4f7ba5b55cb311308b8881764d61c","source":{"kind":"arxiv","id":"2406.15209","version":1},"attestation_state":"computed","paper":{"title":"Prompting Whisper for QA-driven Zero-shot End-to-end Spoken Language Understanding","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"eess.AS","authors_text":"Mohan Li, Rama Doddipatla, Simon Keizer","submitted_at":"2024-06-21T14:51:32Z","abstract_excerpt":"Zero-shot spoken language understanding (SLU) enables systems to comprehend user utterances in new domains without prior exposure to training data. Recent studies often rely on large language models (LLMs), leading to excessive footprints and complexity. This paper proposes the use of Whisper, a standalone speech processing model, for zero-shot end-to-end (E2E) SLU. To handle unseen semantic labels, SLU tasks are integrated into a question-answering (QA) framework, which prompts the Whisper decoder for semantics deduction. The system is efficiently trained with prefix-tuning, optimising a mini"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.15209","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"eess.AS","submitted_at":"2024-06-21T14:51:32Z","cross_cats_sorted":[],"title_canon_sha256":"e8491b4d3ffcfbb16d671228dad9ea75624ea7126bcfbf3cf8b4b27501e243c7","abstract_canon_sha256":"8a3dc805aa4af344600e623c4f87a61b93ecfa143840d9893208a0db51056741"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:35:15.566490Z","signature_b64":"1exDuL42LaKNMlG6Hmu/efQNomtWSomT97VdHjnL82SGoV0PcvKWuWWGcTxydlv57xR6IOLPv38yUv1a88BABw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"87e5d892514246691e62dd3f4abec99992b4f7ba5b55cb311308b8881764d61c","last_reissued_at":"2026-07-05T08:35:15.565957Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:35:15.565957Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Prompting Whisper for QA-driven Zero-shot End-to-end Spoken Language Understanding","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"eess.AS","authors_text":"Mohan Li, Rama Doddipatla, Simon Keizer","submitted_at":"2024-06-21T14:51:32Z","abstract_excerpt":"Zero-shot spoken language understanding (SLU) enables systems to comprehend user utterances in new domains without prior exposure to training data. Recent studies often rely on large language models (LLMs), leading to excessive footprints and complexity. This paper proposes the use of Whisper, a standalone speech processing model, for zero-shot end-to-end (E2E) SLU. To handle unseen semantic labels, SLU tasks are integrated into a question-answering (QA) framework, which prompts the Whisper decoder for semantics deduction. The system is efficiently trained with prefix-tuning, optimising a mini"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.15209","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.15209/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.15209","created_at":"2026-07-05T08:35:15.566032+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.15209v1","created_at":"2026-07-05T08:35:15.566032+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.15209","created_at":"2026-07-05T08:35:15.566032+00:00"},{"alias_kind":"pith_short_12","alias_value":"Q7S5RESRIJDG","created_at":"2026-07-05T08:35:15.566032+00:00"},{"alias_kind":"pith_short_16","alias_value":"Q7S5RESRIJDGSHTC","created_at":"2026-07-05T08:35:15.566032+00:00"},{"alias_kind":"pith_short_8","alias_value":"Q7S5RESR","created_at":"2026-07-05T08:35:15.566032+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.04473","citing_title":"SpeechLLM: Unified Speech and Language Model for Enhanced Multi-Task Understanding in Low Resource Settings","ref_index":19,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Q7S5RESRIJDGSHTC3U7UVPWJTG","json":"https://pith.science/pith/Q7S5RESRIJDGSHTC3U7UVPWJTG.json","graph_json":"https://pith.science/api/pith-number/Q7S5RESRIJDGSHTC3U7UVPWJTG/graph.json","events_json":"https://pith.science/api/pith-number/Q7S5RESRIJDGSHTC3U7UVPWJTG/events.json","paper":"https://pith.science/paper/Q7S5RESR"},"agent_actions":{"view_html":"https://pith.science/pith/Q7S5RESRIJDGSHTC3U7UVPWJTG","download_json":"https://pith.science/pith/Q7S5RESRIJDGSHTC3U7UVPWJTG.json","view_paper":"https://pith.science/paper/Q7S5RESR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.15209&json=true","fetch_graph":"https://pith.science/api/pith-number/Q7S5RESRIJDGSHTC3U7UVPWJTG/graph.json","fetch_events":"https://pith.science/api/pith-number/Q7S5RESRIJDGSHTC3U7UVPWJTG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Q7S5RESRIJDGSHTC3U7UVPWJTG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Q7S5RESRIJDGSHTC3U7UVPWJTG/action/storage_attestation","attest_author":"https://pith.science/pith/Q7S5RESRIJDGSHTC3U7UVPWJTG/action/author_attestation","sign_citation":"https://pith.science/pith/Q7S5RESRIJDGSHTC3U7UVPWJTG/action/citation_signature","submit_replication":"https://pith.science/pith/Q7S5RESRIJDGSHTC3U7UVPWJTG/action/replication_record"}},"created_at":"2026-07-05T08:35:15.566032+00:00","updated_at":"2026-07-05T08:35:15.566032+00:00"}