{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:RLM74DQBKHYFSXBXOXMOE66PVW","short_pith_number":"pith:RLM74DQB","schema_version":"1.0","canonical_sha256":"8ad9fe0e0151f0595c3775d8e27bcfadbbf78edc36dbccfe8d9584b0a5722e0d","source":{"kind":"arxiv","id":"2305.09636","version":1},"attestation_state":"computed","paper":{"title":"SoundStorm: Efficient Parallel Audio Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"Damien Vincent, Eugene Kharitonov, Marco Tagliasacchi, Matt Sharifi, Neil Zeghidour, Zal\\'an Borsos","submitted_at":"2023-05-16T17:41:25Z","abstract_excerpt":"We present SoundStorm, a model for efficient, non-autoregressive audio generation. SoundStorm receives as input the semantic tokens of AudioLM, and relies on bidirectional attention and confidence-based parallel decoding to generate the tokens of a neural audio codec. Compared to the autoregressive generation approach of AudioLM, our model produces audio of the same quality and with higher consistency in voice and acoustic conditions, while being two orders of magnitude faster. SoundStorm generates 30 seconds of audio in 0.5 seconds on a TPU-v4. We demonstrate the ability of our model to scale"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.09636","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2023-05-16T17:41:25Z","cross_cats_sorted":["cs.LG","eess.AS"],"title_canon_sha256":"7c5259e270a2aec6195cf7862ecedddd9f696a8653b76079b59196b3e4e580f2","abstract_canon_sha256":"bd0939826619455ef912c4485a5f27fa61d9177cbe7c1d41f56689c00ba1a530"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:10:50.231354Z","signature_b64":"doj7HPw9j3d5IBCAPWveaubB3cv37TRphRDjc7omUL7Hsngqpn0fJ5aSAejDvamO7S15g9nktu1L/jZhY/XECg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8ad9fe0e0151f0595c3775d8e27bcfadbbf78edc36dbccfe8d9584b0a5722e0d","last_reissued_at":"2026-07-05T06:10:50.230943Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:10:50.230943Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SoundStorm: Efficient Parallel Audio Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"Damien Vincent, Eugene Kharitonov, Marco Tagliasacchi, Matt Sharifi, Neil Zeghidour, Zal\\'an Borsos","submitted_at":"2023-05-16T17:41:25Z","abstract_excerpt":"We present SoundStorm, a model for efficient, non-autoregressive audio generation. SoundStorm receives as input the semantic tokens of AudioLM, and relies on bidirectional attention and confidence-based parallel decoding to generate the tokens of a neural audio codec. Compared to the autoregressive generation approach of AudioLM, our model produces audio of the same quality and with higher consistency in voice and acoustic conditions, while being two orders of magnitude faster. SoundStorm generates 30 seconds of audio in 0.5 seconds on a TPU-v4. We demonstrate the ability of our model to scale"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.09636","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.09636/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.09636","created_at":"2026-07-05T06:10:50.230999+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.09636v1","created_at":"2026-07-05T06:10:50.230999+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.09636","created_at":"2026-07-05T06:10:50.230999+00:00"},{"alias_kind":"pith_short_12","alias_value":"RLM74DQBKHYF","created_at":"2026-07-05T06:10:50.230999+00:00"},{"alias_kind":"pith_short_16","alias_value":"RLM74DQBKHYFSXBX","created_at":"2026-07-05T06:10:50.230999+00:00"},{"alias_kind":"pith_short_8","alias_value":"RLM74DQB","created_at":"2026-07-05T06:10:50.230999+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":21,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06027","citing_title":"Fr\\'echet Distance Loss on Speech Representations for Text-to-Speech Synthesis","ref_index":26,"is_internal_anchor":true},{"citing_arxiv_id":"2606.27320","citing_title":"Elastic Time: Dynamic Frame Rate Bottlenecks for Neural Audio Coding","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09098","citing_title":"HoliDubber: Holistic Video Dubbing for Complex Acoustic Scenes via Text-Guided Audio Synthesis","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06928","citing_title":"VoxCPM2 Technical Report","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31247","citing_title":"FlexiSLM: A Dynamic and Controllable Frame Rate Spoken Language Model","ref_index":133,"is_internal_anchor":false},{"citing_arxiv_id":"2504.08528","citing_title":"On The Landscape of Spoken Language Models: A Comprehensive Survey","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17085","citing_title":"Taming Audio VAEs via Target-KL Regularization","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2505.24437","citing_title":"SwitchCodec: A High-Fidelity Nerual Audio Codec With Sparse Quantization","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2509.19883","citing_title":"CoMelSinger: Discrete Token-Based Zero-Shot Singing Synthesis With Structured Melody Control and Guidance","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2510.06201","citing_title":"TokenChain: A Discrete Speech Chain via Semantic Token Modeling","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2602.02494","citing_title":"MEG-XL: Data-Efficient Brain-to-Text via Long-Context Pre-Training","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2306.12925","citing_title":"AudioPaLM: A Large Language Model That Can Speak and Listen","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14555","citing_title":"Break-the-Beat! Controllable MIDI-to-Drum Audio Synthesis","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.00688","citing_title":"OmniVoice: Towards Omnilingual Zero-Shot Text-to-Speech with Diffusion Language Models","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2410.00037","citing_title":"Moshi: a speech-text foundation model for real-time dialogue","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09971","citing_title":"HapticLDM: A Diffusion Model for Text-to-Vibrotactile Generation","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09386","citing_title":"Kinetic-Optimal Scheduling with Moment Correction for Metric-Induced Discrete Flow Matching in Zero-Shot Text-to-Speech","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00329","citing_title":"Fast Text-to-Audio Generation with One-Step Sampling via Energy-Scoring and Auxiliary Contextual Representation Distillation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17986","citing_title":"Latent Fourier Transform","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11052","citing_title":"LaDA-Band: Language Diffusion Models for Vocal-to-Accompaniment Generation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12456","citing_title":"X-VC: Zero-shot Streaming Voice Conversion in Codec Space","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RLM74DQBKHYFSXBXOXMOE66PVW","json":"https://pith.science/pith/RLM74DQBKHYFSXBXOXMOE66PVW.json","graph_json":"https://pith.science/api/pith-number/RLM74DQBKHYFSXBXOXMOE66PVW/graph.json","events_json":"https://pith.science/api/pith-number/RLM74DQBKHYFSXBXOXMOE66PVW/events.json","paper":"https://pith.science/paper/RLM74DQB"},"agent_actions":{"view_html":"https://pith.science/pith/RLM74DQBKHYFSXBXOXMOE66PVW","download_json":"https://pith.science/pith/RLM74DQBKHYFSXBXOXMOE66PVW.json","view_paper":"https://pith.science/paper/RLM74DQB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.09636&json=true","fetch_graph":"https://pith.science/api/pith-number/RLM74DQBKHYFSXBXOXMOE66PVW/graph.json","fetch_events":"https://pith.science/api/pith-number/RLM74DQBKHYFSXBXOXMOE66PVW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RLM74DQBKHYFSXBXOXMOE66PVW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RLM74DQBKHYFSXBXOXMOE66PVW/action/storage_attestation","attest_author":"https://pith.science/pith/RLM74DQBKHYFSXBXOXMOE66PVW/action/author_attestation","sign_citation":"https://pith.science/pith/RLM74DQBKHYFSXBXOXMOE66PVW/action/citation_signature","submit_replication":"https://pith.science/pith/RLM74DQBKHYFSXBXOXMOE66PVW/action/replication_record"}},"created_at":"2026-07-05T06:10:50.230999+00:00","updated_at":"2026-07-05T06:10:50.230999+00:00"}