{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:CFDW2DEIBMGDIV55D2HMVKQRK7","short_pith_number":"pith:CFDW2DEI","schema_version":"1.0","canonical_sha256":"11476d0c880b0c3457bd1e8ecaaa1157c9fc1931a84a003e62be4d88e8671ec0","source":{"kind":"arxiv","id":"2310.00704","version":6},"attestation_state":"computed","paper":{"title":"UniAudio: An Audio Foundation Model Toward Universal Audio Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.SD","authors_text":"Dongchao Yang, Helen Meng, Jiang Bian, Jiatong Shi, Jinchuan Tian, Rongjie Huang, Sheng Zhao, Songxiang Liu, Xixin Wu, Xuankai Chang, Xu Tan, Zhou Zhao","submitted_at":"2023-10-01T15:49:46Z","abstract_excerpt":"Large Language models (LLM) have demonstrated the capability to handle a variety of generative tasks. This paper presents the UniAudio system, which, unlike prior task-specific approaches, leverages LLM techniques to generate multiple types of audio (including speech, sounds, music, and singing) with given input conditions. UniAudio 1) first tokenizes all types of target audio along with other condition modalities, 2) concatenates source-target pair as a single sequence, and 3) performs next-token prediction using LLM. Also, a multi-scale Transformer model is proposed to handle the overly long"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.00704","kind":"arxiv","version":6},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2023-10-01T15:49:46Z","cross_cats_sorted":["eess.AS"],"title_canon_sha256":"22a29d5ac28ee0ddcf4fa8179e3cc1b05d2ccb9a4de7501cdad39b2daa9975e6","abstract_canon_sha256":"95cd717763bb9ca42cd7f3c0fbfe285eb25a68d9185ba7fb2e5418561b885141"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:46:43.112716Z","signature_b64":"Qc0VxjpHU74GmgxvSUksM00ZJA4tYlsiBDXgRkKKgKAsqljChGyBh2YzF/7f0Ufyc/YFEP6eFiQ+XJCOGKcqBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"11476d0c880b0c3457bd1e8ecaaa1157c9fc1931a84a003e62be4d88e8671ec0","last_reissued_at":"2026-07-05T09:46:43.112162Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:46:43.112162Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"UniAudio: An Audio Foundation Model Toward Universal Audio Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.SD","authors_text":"Dongchao Yang, Helen Meng, Jiang Bian, Jiatong Shi, Jinchuan Tian, Rongjie Huang, Sheng Zhao, Songxiang Liu, Xixin Wu, Xuankai Chang, Xu Tan, Zhou Zhao","submitted_at":"2023-10-01T15:49:46Z","abstract_excerpt":"Large Language models (LLM) have demonstrated the capability to handle a variety of generative tasks. This paper presents the UniAudio system, which, unlike prior task-specific approaches, leverages LLM techniques to generate multiple types of audio (including speech, sounds, music, and singing) with given input conditions. UniAudio 1) first tokenizes all types of target audio along with other condition modalities, 2) concatenates source-target pair as a single sequence, and 3) performs next-token prediction using LLM. Also, a multi-scale Transformer model is proposed to handle the overly long"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.00704","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.00704/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.00704","created_at":"2026-07-05T09:46:43.112222+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.00704v6","created_at":"2026-07-05T09:46:43.112222+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.00704","created_at":"2026-07-05T09:46:43.112222+00:00"},{"alias_kind":"pith_short_12","alias_value":"CFDW2DEIBMGD","created_at":"2026-07-05T09:46:43.112222+00:00"},{"alias_kind":"pith_short_16","alias_value":"CFDW2DEIBMGDIV55","created_at":"2026-07-05T09:46:43.112222+00:00"},{"alias_kind":"pith_short_8","alias_value":"CFDW2DEI","created_at":"2026-07-05T09:46:43.112222+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05196","citing_title":"Unified Audio Intelligence Without Regressing on Text Intelligence","ref_index":49,"is_internal_anchor":true},{"citing_arxiv_id":"2606.23080","citing_title":"AudioCALM: Continuous Autoregressive Language Modeling for Universal Audio Generation","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21372","citing_title":"NAC: Neural Action Codec for Vision-Language-Action Models","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12940","citing_title":"Self-Guidance: Enhancing Neural Codecs via Decoder Manifold Alignment","ref_index":134,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01677","citing_title":"UniVocal: Unified Speech-Singing Code-Switching Synthesis","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28063","citing_title":"Unified Synthesis of Compositional Speech and Sound from Free-Form Text Prompts","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2509.03526","citing_title":"Enhancing Speech Large Language Models through Reinforced Behavior Alignment","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2510.13293","citing_title":"Cross-modal Consistency Guidance for Robust Emotion Control in Auto-Regressive TTS Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15831","citing_title":"Modeling Music as a Time-Frequency Image: A 2D Tokenizer for Music Generation","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2510.13293","citing_title":"Cross-modal Consistency Guidance for Robust Emotion Control in Auto-Regressive TTS Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2601.15621","citing_title":"Qwen3-TTS Technical Report","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2410.00037","citing_title":"Moshi: a speech-text foundation model for real-time dialogue","ref_index":105,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09259","citing_title":"Remix the Timbre: Diffusion-Based Style Transfer Across Polyphonic Stems","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2504.18425","citing_title":"Kimi-Audio Technical Report","ref_index":77,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CFDW2DEIBMGDIV55D2HMVKQRK7","json":"https://pith.science/pith/CFDW2DEIBMGDIV55D2HMVKQRK7.json","graph_json":"https://pith.science/api/pith-number/CFDW2DEIBMGDIV55D2HMVKQRK7/graph.json","events_json":"https://pith.science/api/pith-number/CFDW2DEIBMGDIV55D2HMVKQRK7/events.json","paper":"https://pith.science/paper/CFDW2DEI"},"agent_actions":{"view_html":"https://pith.science/pith/CFDW2DEIBMGDIV55D2HMVKQRK7","download_json":"https://pith.science/pith/CFDW2DEIBMGDIV55D2HMVKQRK7.json","view_paper":"https://pith.science/paper/CFDW2DEI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.00704&json=true","fetch_graph":"https://pith.science/api/pith-number/CFDW2DEIBMGDIV55D2HMVKQRK7/graph.json","fetch_events":"https://pith.science/api/pith-number/CFDW2DEIBMGDIV55D2HMVKQRK7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CFDW2DEIBMGDIV55D2HMVKQRK7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CFDW2DEIBMGDIV55D2HMVKQRK7/action/storage_attestation","attest_author":"https://pith.science/pith/CFDW2DEIBMGDIV55D2HMVKQRK7/action/author_attestation","sign_citation":"https://pith.science/pith/CFDW2DEIBMGDIV55D2HMVKQRK7/action/citation_signature","submit_replication":"https://pith.science/pith/CFDW2DEIBMGDIV55D2HMVKQRK7/action/replication_record"}},"created_at":"2026-07-05T09:46:43.112222+00:00","updated_at":"2026-07-05T09:46:43.112222+00:00"}