{"work":{"id":"f51f1d58-4625-4dca-a20f-73a7705ac847","openalex_id":null,"doi":null,"arxiv_id":"2407.14358","raw_key":null,"title":"Stable Audio Open","authors":null,"authors_text":"Z","year":2024,"venue":"cs.SD","abstract":"Open generative models are vitally important for the community, allowing for fine-tunes and serving as baselines when presenting new models. However, most current text-to-audio models are private and not accessible for artists and researchers to build upon. Here we describe the architecture and training process of a new open-weights text-to-audio model trained with Creative Commons data. Our evaluation shows that the model's performance is competitive with the state-of-the-art across various metrics. Notably, the reported FDopenl3 results (measuring the realism of the generations) showcase its potential for high-quality stereo sound synthesis at 44.1kHz.","external_url":"https://arxiv.org/abs/2407.14358","cited_by_count":null,"metadata_source":"pith","metadata_fetched_at":"2026-07-08T00:04:22.330619+00:00","pith_arxiv_id":"2407.14358","created_at":"2026-05-10T07:26:59.761103+00:00","updated_at":"2026-07-08T00:04:22.330619+00:00","title_quality_ok":false,"display_title":"D., Carr, C","render_title":"D., Carr, C"},"hub":{"state":{"work_id":"f51f1d58-4625-4dca-a20f-73a7705ac847","tier":"hub","tier_reason":"10+ Pith inbound or 1,000+ external citations","pith_inbound_count":17,"external_cited_by_count":null,"distinct_field_count":6,"first_pith_cited_at":"2024-09-17T17:55:39+00:00","last_pith_cited_at":"2026-07-06T15:11:57+00:00","author_build_status":"not_needed","summary_status":"needed","contexts_status":"needed","graph_status":"needed","ask_index_status":"not_needed","reader_status":"not_needed","recognition_status":"not_needed","updated_at":"2026-08-22T05:29:45.195653+00:00","tier_text":"hub"},"tier":"hub","role_counts":[{"context_role":"background","n":1}],"polarity_counts":[{"context_polarity":"background","n":1}],"runs":{},"summary":{},"graph":{},"authors":[]}}