{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2018:7AD7MQAPB3V3NXTJ7J4MVGBTPF","short_pith_number":"pith:7AD7MQAP","schema_version":"1.0","canonical_sha256":"f807f6400f0eebb6de69fa78ca98337975cd34b9c03188de0b3da18f6e75cf95","source":{"kind":"arxiv","id":"1804.03209","version":1},"attestation_state":"computed","paper":{"title":"Speech Commands: A Dataset for Limited-Vocabulary Speech Recognition","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.HC"],"primary_cat":"cs.CL","authors_text":"Pete Warden","submitted_at":"2018-04-09T19:58:17Z","abstract_excerpt":"Describes an audio dataset of spoken words designed to help train and evaluate keyword spotting systems. Discusses why this task is an interesting challenge, and why it requires a specialized dataset that is different from conventional datasets used for automatic speech recognition of full sentences. Suggests a methodology for reproducible and comparable accuracy metrics for this task. Describes how the data was collected and verified, what it contains, previous versions and properties. Concludes by reporting baseline results of models trained on this dataset."},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1804.03209","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2018-04-09T19:58:17Z","cross_cats_sorted":["cs.HC"],"title_canon_sha256":"3c0180c17c3d030628665ec787e985e91d40936a2938132475372aad51212b38","abstract_canon_sha256":"6445e35d2768884115e3e1ebea91e7a7f0bae3181d3f7be3c3a8d492cc0ccff4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T00:18:56.698951Z","signature_b64":"gM6qDDqm0O+1rdg4pGi630JlDrEAjjmzDvP43yhtC9vN6N9BVFU6Z0M1sPX609aC/yAzH1D1EQUivL7/MxWrAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f807f6400f0eebb6de69fa78ca98337975cd34b9c03188de0b3da18f6e75cf95","last_reissued_at":"2026-05-18T00:18:56.698499Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T00:18:56.698499Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Speech Commands: A Dataset for Limited-Vocabulary Speech Recognition","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.HC"],"primary_cat":"cs.CL","authors_text":"Pete Warden","submitted_at":"2018-04-09T19:58:17Z","abstract_excerpt":"Describes an audio dataset of spoken words designed to help train and evaluate keyword spotting systems. Discusses why this task is an interesting challenge, and why it requires a specialized dataset that is different from conventional datasets used for automatic speech recognition of full sentences. Suggests a methodology for reproducible and comparable accuracy metrics for this task. Describes how the data was collected and verified, what it contains, previous versions and properties. Concludes by reporting baseline results of models trained on this dataset."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1804.03209","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1804.03209","created_at":"2026-05-18T00:18:56.698578+00:00"},{"alias_kind":"arxiv_version","alias_value":"1804.03209v1","created_at":"2026-05-18T00:18:56.698578+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1804.03209","created_at":"2026-05-18T00:18:56.698578+00:00"},{"alias_kind":"pith_short_12","alias_value":"7AD7MQAPB3V3","created_at":"2026-05-18T12:32:11.075285+00:00"},{"alias_kind":"pith_short_16","alias_value":"7AD7MQAPB3V3NXTJ","created_at":"2026-05-18T12:32:11.075285+00:00"},{"alias_kind":"pith_short_8","alias_value":"7AD7MQAP","created_at":"2026-05-18T12:32:11.075285+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":58,"internal_anchor_count":44,"sample":[{"citing_arxiv_id":"2607.07907","citing_title":"Multimodal Unlearning Across Vision, Language, Video, and Audio: Survey of Methods, Datasets, and Benchmarks","ref_index":256,"is_internal_anchor":true},{"citing_arxiv_id":"2607.08427","citing_title":"FPGN: Redefining Ultra-Fast Programmable Gate-based Neural Acceleration with Differentiable LUTs","ref_index":39,"is_internal_anchor":true},{"citing_arxiv_id":"2607.05250","citing_title":"Streaming Neural Speech Codecs through Time-Invariant Representations","ref_index":14,"is_internal_anchor":true},{"citing_arxiv_id":"2606.26556","citing_title":"WQ-Fusion: Dynamic Gated Attention for Cross-Domain Audio Representation","ref_index":32,"is_internal_anchor":true},{"citing_arxiv_id":"2606.24910","citing_title":"End-to-End Voice Intent Recognition for Spontaneous Human-Drone Interaction with Naive Users","ref_index":19,"is_internal_anchor":true},{"citing_arxiv_id":"2606.20893","citing_title":"Exploiting Neural Audio Codec Latents for Adversarial Audio Attacks","ref_index":33,"is_internal_anchor":true},{"citing_arxiv_id":"2606.20106","citing_title":"Personalized Keyword Spotting for User-Defined Keywords Leveraging Text-Independent Speaker Verification","ref_index":32,"is_internal_anchor":true},{"citing_arxiv_id":"2606.19039","citing_title":"Adaptive Speech-to-Spike Encoding for Spiking Neural Networks","ref_index":21,"is_internal_anchor":true},{"citing_arxiv_id":"2606.18664","citing_title":"NeuralMUSIC: A Hybrid Neural-Subspace Framework for Robot Sound Source Localization","ref_index":16,"is_internal_anchor":true},{"citing_arxiv_id":"2606.18122","citing_title":"Embedded Machine Learning for Microcontroller-Class Edge Devices: Data, Feature, Evaluation, and Deployment Pipelines","ref_index":5,"is_internal_anchor":true},{"citing_arxiv_id":"2606.17967","citing_title":"Learning task-specific subspaces via interventional post-training of speech foundation models","ref_index":41,"is_internal_anchor":true},{"citing_arxiv_id":"2606.13379","citing_title":"Positional Encoding in the Context of Memristor-Based Analog Computation for Automatic Speech Recognition","ref_index":34,"is_internal_anchor":true},{"citing_arxiv_id":"2607.01729","citing_title":"DRL-CLBA: A Clean Label Backdoor Attack for Speech Classification via DDPG Reinforcement Learning","ref_index":39,"is_internal_anchor":true},{"citing_arxiv_id":"2606.12895","citing_title":"LongSpike: Fractional Order Spiking State Space Models for Efficient Long Sequence Learning","ref_index":53,"is_internal_anchor":true},{"citing_arxiv_id":"2607.01702","citing_title":"Pmeta-TLA: Backdoor Attacks for Speech Classification Models via Meta-Learning with Timbre Leakage Attack","ref_index":56,"is_internal_anchor":true},{"citing_arxiv_id":"2606.12059","citing_title":"Attention by Synchronization in Coupled Oscillator Networks","ref_index":7,"is_internal_anchor":true},{"citing_arxiv_id":"2606.04210","citing_title":"Representation Matters in Randomized Smoothing for Audio Classification","ref_index":7,"is_internal_anchor":true},{"citing_arxiv_id":"2606.02631","citing_title":"Wavelet as Tokenizer: Preliminary Results on a Shared Wavelet Token Schema for Natural Signals","ref_index":19,"is_internal_anchor":true},{"citing_arxiv_id":"2605.26323","citing_title":"Totoro$^+$: An Adaptive and Scalable Edge Federated Learning System","ref_index":77,"is_internal_anchor":true},{"citing_arxiv_id":"2605.11855","citing_title":"Improving the Performance and Learning Stability of Parallelizable RNNs Designed for Ultra-Low Power Applications","ref_index":6,"is_internal_anchor":true},{"citing_arxiv_id":"2605.15216","citing_title":"Hardware-Software Co-Design of Scalable, Energy-Efficient Analog Recurrent Computations","ref_index":62,"is_internal_anchor":true},{"citing_arxiv_id":"2606.18664","citing_title":"NeuralMUSIC: A Hybrid Neural-Subspace Framework for Robot Sound Source Localization","ref_index":16,"is_internal_anchor":true},{"citing_arxiv_id":"2606.28953","citing_title":"Clustering Unsupervised Representations as Defense against Poisoning Attacks on Speech Commands Classification System","ref_index":50,"is_internal_anchor":true},{"citing_arxiv_id":"2605.26848","citing_title":"Design principles for optoelectronic light-scattering reservoir computing at the edge of chaos","ref_index":6,"is_internal_anchor":true},{"citing_arxiv_id":"2605.31226","citing_title":"What changes after deployment? A survey on On-device Learning in TinyML","ref_index":104,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7AD7MQAPB3V3NXTJ7J4MVGBTPF","json":"https://pith.science/pith/7AD7MQAPB3V3NXTJ7J4MVGBTPF.json","graph_json":"https://pith.science/api/pith-number/7AD7MQAPB3V3NXTJ7J4MVGBTPF/graph.json","events_json":"https://pith.science/api/pith-number/7AD7MQAPB3V3NXTJ7J4MVGBTPF/events.json","paper":"https://pith.science/paper/7AD7MQAP"},"agent_actions":{"view_html":"https://pith.science/pith/7AD7MQAPB3V3NXTJ7J4MVGBTPF","download_json":"https://pith.science/pith/7AD7MQAPB3V3NXTJ7J4MVGBTPF.json","view_paper":"https://pith.science/paper/7AD7MQAP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1804.03209&json=true","fetch_graph":"https://pith.science/api/pith-number/7AD7MQAPB3V3NXTJ7J4MVGBTPF/graph.json","fetch_events":"https://pith.science/api/pith-number/7AD7MQAPB3V3NXTJ7J4MVGBTPF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7AD7MQAPB3V3NXTJ7J4MVGBTPF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7AD7MQAPB3V3NXTJ7J4MVGBTPF/action/storage_attestation","attest_author":"https://pith.science/pith/7AD7MQAPB3V3NXTJ7J4MVGBTPF/action/author_attestation","sign_citation":"https://pith.science/pith/7AD7MQAPB3V3NXTJ7J4MVGBTPF/action/citation_signature","submit_replication":"https://pith.science/pith/7AD7MQAPB3V3NXTJ7J4MVGBTPF/action/replication_record"}},"created_at":"2026-05-18T00:18:56.698578+00:00","updated_at":"2026-05-18T00:18:56.698578+00:00"}