{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:6A3RYLSTAY2COGNFEAY5YW35GF","short_pith_number":"pith:6A3RYLST","schema_version":"1.0","canonical_sha256":"f0371c2e5306342719a52031dc5b7d3141f504793bcc11795247a8368398aa69","source":{"kind":"arxiv","id":"2312.15821","version":1},"attestation_state":"computed","paper":{"title":"Audiobox: Unified Audio Generation with Natural Language Prompts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"Akinniyi Akinyemi, Alice Rakotoarison, Andros Tjandra, Apoorv Vyas, Baishan Guo, Bapi Akula, Bowen Shi, Brian Ellis, Carleigh Wood, Chris Summers, Ivan Cruz, Jeff Wang, Jiemin Zhang, Joshua Lane, Liang Tan, Mary Williamson, Matthew Le, Rashel Moritz, Robert Adkins, Wei-Ning Hsu, William Ngan, Xinyue Zhang, Yael Yungster, Yi-Chiao Wu","submitted_at":"2023-12-25T22:24:49Z","abstract_excerpt":"Audio is an essential part of our life, but creating it often requires expertise and is time-consuming. Research communities have made great progress over the past year advancing the performance of large scale audio generative models for a single modality (speech, sound, or music) through adopting more powerful generative models and scaling data. However, these models lack controllability in several aspects: speech generation models cannot synthesize novel styles based on text description and are limited on domain coverage such as outdoor environments; sound generation models only provide coar"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.15821","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2023-12-25T22:24:49Z","cross_cats_sorted":["cs.LG","eess.AS"],"title_canon_sha256":"79057c69eb613c6d534d2b44557858fca9659bf1ffce97257c61d35985d14e57","abstract_canon_sha256":"70effce4d16e604c4f4edf6eabde4eed54aba8c4420658eeb41d15f86f8116ba"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:28:01.880691Z","signature_b64":"prPjrD8wPoAq3OZ2Gnv/zqqqsBIxEO+wPnI4NY5lto1+5QTkTrQuz3E0qdCXqb++7nQiBz/qVJMj0f7F+QJNCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f0371c2e5306342719a52031dc5b7d3141f504793bcc11795247a8368398aa69","last_reissued_at":"2026-07-05T07:28:01.880184Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:28:01.880184Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Audiobox: Unified Audio Generation with Natural Language Prompts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"Akinniyi Akinyemi, Alice Rakotoarison, Andros Tjandra, Apoorv Vyas, Baishan Guo, Bapi Akula, Bowen Shi, Brian Ellis, Carleigh Wood, Chris Summers, Ivan Cruz, Jeff Wang, Jiemin Zhang, Joshua Lane, Liang Tan, Mary Williamson, Matthew Le, Rashel Moritz, Robert Adkins, Wei-Ning Hsu, William Ngan, Xinyue Zhang, Yael Yungster, Yi-Chiao Wu","submitted_at":"2023-12-25T22:24:49Z","abstract_excerpt":"Audio is an essential part of our life, but creating it often requires expertise and is time-consuming. Research communities have made great progress over the past year advancing the performance of large scale audio generative models for a single modality (speech, sound, or music) through adopting more powerful generative models and scaling data. However, these models lack controllability in several aspects: speech generation models cannot synthesize novel styles based on text description and are limited on domain coverage such as outdoor environments; sound generation models only provide coar"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.15821","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.15821/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.15821","created_at":"2026-07-05T07:28:01.880245+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.15821v1","created_at":"2026-07-05T07:28:01.880245+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.15821","created_at":"2026-07-05T07:28:01.880245+00:00"},{"alias_kind":"pith_short_12","alias_value":"6A3RYLSTAY2C","created_at":"2026-07-05T07:28:01.880245+00:00"},{"alias_kind":"pith_short_16","alias_value":"6A3RYLSTAY2COGNF","created_at":"2026-07-05T07:28:01.880245+00:00"},{"alias_kind":"pith_short_8","alias_value":"6A3RYLST","created_at":"2026-07-05T07:28:01.880245+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":22,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05196","citing_title":"Unified Audio Intelligence Without Regressing on Text Intelligence","ref_index":45,"is_internal_anchor":true},{"citing_arxiv_id":"2606.23080","citing_title":"AudioCALM: Continuous Autoregressive Language Modeling for Universal Audio Generation","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09098","citing_title":"HoliDubber: Holistic Video Dubbing for Complex Acoustic Scenes via Text-Guided Audio Synthesis","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06928","citing_title":"VoxCPM2 Technical Report","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30965","citing_title":"ImmersiveTTS: Environment-Aware Text-to-Speech with Multimodal Diffusion Transformer and Domain-Specific Representation Alignment","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31729","citing_title":"Is Natural Always Appropriate? Investigating Naturalness and Appropriateness Across Different Domains for TTS Evaluation","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08729","citing_title":"Unison: Harmonizing Motion, Speech, and Sound for Human-Centric Audio-Video Generation","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24652","citing_title":"AVBench: Human-Aligned and Automated Evaluation Benchmark for Audio-Video Generative Models","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28063","citing_title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31530","citing_title":"UNISON: A Unified Sound Generation and Editing Framework via Deep LLM Fusion","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02739","citing_title":"EntangleCodec: A Unified Discrete Audio Tokenizer via Semantic-Acoustic Entanglement","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17488","citing_title":"Omni-Customizer: End-to-End MultiModal Customization for Joint Audio-Video Generation","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2509.06027","citing_title":"DreamAudio: Customized Text-to-Audio Generation with Diffusion Models","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2502.05139","citing_title":"Meta Audiobox Aesthetics: Unified Automatic Quality Assessment for Speech, Music, and Sound","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08580","citing_title":"Adjoint Matching through the Lens of the Stochastic Maximum Principle in Optimal Control","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2412.06264","citing_title":"Flow Matching Guide and Code","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08729","citing_title":"Unison: Harmonizing Motion, Speech, and Sound for Human-Centric Audio-Video Generation","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00329","citing_title":"Fast Text-to-Audio Generation with One-Step Sampling via Energy-Scoring and Auxiliary Contextual Representation Distillation","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00229","citing_title":"A unified perspective on fine-tuning and sampling with diffusion and flow models","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2410.13720","citing_title":"Movie Gen: A Cast of Media Foundation Models","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09111","citing_title":"PS-TTS: Phonetic Synchronization in Text-to-Speech for Achieving Natural Automated Dubbing","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05526","citing_title":"Controllable Singing Style Conversion with Boundary-Aware Information Bottleneck","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6A3RYLSTAY2COGNFEAY5YW35GF","json":"https://pith.science/pith/6A3RYLSTAY2COGNFEAY5YW35GF.json","graph_json":"https://pith.science/api/pith-number/6A3RYLSTAY2COGNFEAY5YW35GF/graph.json","events_json":"https://pith.science/api/pith-number/6A3RYLSTAY2COGNFEAY5YW35GF/events.json","paper":"https://pith.science/paper/6A3RYLST"},"agent_actions":{"view_html":"https://pith.science/pith/6A3RYLSTAY2COGNFEAY5YW35GF","download_json":"https://pith.science/pith/6A3RYLSTAY2COGNFEAY5YW35GF.json","view_paper":"https://pith.science/paper/6A3RYLST","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.15821&json=true","fetch_graph":"https://pith.science/api/pith-number/6A3RYLSTAY2COGNFEAY5YW35GF/graph.json","fetch_events":"https://pith.science/api/pith-number/6A3RYLSTAY2COGNFEAY5YW35GF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6A3RYLSTAY2COGNFEAY5YW35GF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6A3RYLSTAY2COGNFEAY5YW35GF/action/storage_attestation","attest_author":"https://pith.science/pith/6A3RYLSTAY2COGNFEAY5YW35GF/action/author_attestation","sign_citation":"https://pith.science/pith/6A3RYLSTAY2COGNFEAY5YW35GF/action/citation_signature","submit_replication":"https://pith.science/pith/6A3RYLSTAY2COGNFEAY5YW35GF/action/replication_record"}},"created_at":"2026-07-05T07:28:01.880245+00:00","updated_at":"2026-07-05T07:28:01.880245+00:00"}