{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OXTJAKCNXSSJNUWEV4F7WWHEEU","short_pith_number":"pith:OXTJAKCN","schema_version":"1.0","canonical_sha256":"75e690284dbca496d2c4af0bfb58e4253b251e7cf5c0cfa54ccaa4cb8ff30fa8","source":{"kind":"arxiv","id":"2406.05806","version":4},"attestation_state":"computed","paper":{"title":"Do Prompts Really Prompt? Exploring the Prompt Understanding Capability of Whisper","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Chih-Kai Yang, Hung-yi Lee, Kuan-Po Huang","submitted_at":"2024-06-09T14:44:59Z","abstract_excerpt":"This research explores how the information of prompts interacts with the high-performing speech recognition model, Whisper. We compare its performances when prompted by prompts with correct information and those corrupted with incorrect information. Our results unexpectedly show that Whisper may not understand the textual prompts in a human-expected way. Additionally, we find that performance improvement is not guaranteed even with stronger adherence to the topic information in textual prompts. It is also noted that English prompts generally outperform Mandarin ones on datasets of both languag"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.05806","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-09T14:44:59Z","cross_cats_sorted":["cs.SD","eess.AS"],"title_canon_sha256":"65d0e9e53cac0344d744f543c0a4b85a97a43aa9e59e29a625172529c5b31fab","abstract_canon_sha256":"d78ae89c47e14eeb942917fde44412c265f129bcb9a95b467f370bcc0ed94d8a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:07:29.224620Z","signature_b64":"DKI6bfjFK7mA1uyut3LA4OqtpD6+s2hrArpw3H7ADnySoguD38OP/SJCqV+M1RTzyhjsSLcLkQGV/q2KFAI0CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"75e690284dbca496d2c4af0bfb58e4253b251e7cf5c0cfa54ccaa4cb8ff30fa8","last_reissued_at":"2026-07-05T09:07:29.224169Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:07:29.224169Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Do Prompts Really Prompt? Exploring the Prompt Understanding Capability of Whisper","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Chih-Kai Yang, Hung-yi Lee, Kuan-Po Huang","submitted_at":"2024-06-09T14:44:59Z","abstract_excerpt":"This research explores how the information of prompts interacts with the high-performing speech recognition model, Whisper. We compare its performances when prompted by prompts with correct information and those corrupted with incorrect information. Our results unexpectedly show that Whisper may not understand the textual prompts in a human-expected way. Additionally, we find that performance improvement is not guaranteed even with stronger adherence to the topic information in textual prompts. It is also noted that English prompts generally outperform Mandarin ones on datasets of both languag"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.05806","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.05806/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.05806","created_at":"2026-07-05T09:07:29.224226+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.05806v4","created_at":"2026-07-05T09:07:29.224226+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.05806","created_at":"2026-07-05T09:07:29.224226+00:00"},{"alias_kind":"pith_short_12","alias_value":"OXTJAKCNXSSJ","created_at":"2026-07-05T09:07:29.224226+00:00"},{"alias_kind":"pith_short_16","alias_value":"OXTJAKCNXSSJNUWE","created_at":"2026-07-05T09:07:29.224226+00:00"},{"alias_kind":"pith_short_8","alias_value":"OXTJAKCN","created_at":"2026-07-05T09:07:29.224226+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.16474","citing_title":"Enhancing Multilingual ASR for Unseen Languages via Language Embedding Modeling","ref_index":9,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OXTJAKCNXSSJNUWEV4F7WWHEEU","json":"https://pith.science/pith/OXTJAKCNXSSJNUWEV4F7WWHEEU.json","graph_json":"https://pith.science/api/pith-number/OXTJAKCNXSSJNUWEV4F7WWHEEU/graph.json","events_json":"https://pith.science/api/pith-number/OXTJAKCNXSSJNUWEV4F7WWHEEU/events.json","paper":"https://pith.science/paper/OXTJAKCN"},"agent_actions":{"view_html":"https://pith.science/pith/OXTJAKCNXSSJNUWEV4F7WWHEEU","download_json":"https://pith.science/pith/OXTJAKCNXSSJNUWEV4F7WWHEEU.json","view_paper":"https://pith.science/paper/OXTJAKCN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.05806&json=true","fetch_graph":"https://pith.science/api/pith-number/OXTJAKCNXSSJNUWEV4F7WWHEEU/graph.json","fetch_events":"https://pith.science/api/pith-number/OXTJAKCNXSSJNUWEV4F7WWHEEU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OXTJAKCNXSSJNUWEV4F7WWHEEU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OXTJAKCNXSSJNUWEV4F7WWHEEU/action/storage_attestation","attest_author":"https://pith.science/pith/OXTJAKCNXSSJNUWEV4F7WWHEEU/action/author_attestation","sign_citation":"https://pith.science/pith/OXTJAKCNXSSJNUWEV4F7WWHEEU/action/citation_signature","submit_replication":"https://pith.science/pith/OXTJAKCNXSSJNUWEV4F7WWHEEU/action/replication_record"}},"created_at":"2026-07-05T09:07:29.224226+00:00","updated_at":"2026-07-05T09:07:29.224226+00:00"}