{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:GLREUXZGGM4FZCWEJLSEY53WBK","short_pith_number":"pith:GLREUXZG","schema_version":"1.0","canonical_sha256":"32e24a5f2633385c8ac44ae44c77760a959e6a04795567c07db23925ef60d22f","source":{"kind":"arxiv","id":"2312.09911","version":3},"attestation_state":"computed","paper":{"title":"Amphion: An Open-Source Audio, Music and Speech Generation Toolkit","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.SD","authors_text":"Chaoren Wang, Haizhou Li, Haopeng Chen, Haorui He, Jiaqi Li, Junan Zhang, Jun Han, Kai Chen, Lexiao Zou, Liumeng Xue, Mingxuan Wang, Songting Liu, Tze Ying Tang, Xi Chen, Xueyao Zhang, Yicheng Gu, Yuancheng Wang, Zhizheng Wu, Zihao Fang","submitted_at":"2023-12-15T16:23:21Z","abstract_excerpt":"Amphion is an open-source toolkit for Audio, Music, and Speech Generation, targeting to ease the way for junior researchers and engineers into these fields. It presents a unified framework that includes diverse generation tasks and models, with the added bonus of being easily extendable for new incorporation. The toolkit is designed with beginner-friendly workflows and pre-trained models, allowing both beginners and seasoned researchers to kick-start their projects with relative ease. The initial release of Amphion v0.1 supports a range of tasks including Text to Speech (TTS), Text to Audio (T"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.09911","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.SD","submitted_at":"2023-12-15T16:23:21Z","cross_cats_sorted":["eess.AS"],"title_canon_sha256":"06140e4c009db3cb103b65f76f68589a7ec03aaa4ac9d17a305102f8cbc259b5","abstract_canon_sha256":"afd64d8f7b54be2249092cf02339ce50b4b587a312ac8499d831c94f133495c4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:07:25.597355Z","signature_b64":"8o1+YEWc05MKATYzO1i26cv2u7SHrLwhO6RM62YQr6jqejyPtx+oYfqw+dtbj8WheiZcTjw+XBYVoPN7Pz+6BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"32e24a5f2633385c8ac44ae44c77760a959e6a04795567c07db23925ef60d22f","last_reissued_at":"2026-07-05T09:07:25.596876Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:07:25.596876Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Amphion: An Open-Source Audio, Music and Speech Generation Toolkit","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.SD","authors_text":"Chaoren Wang, Haizhou Li, Haopeng Chen, Haorui He, Jiaqi Li, Junan Zhang, Jun Han, Kai Chen, Lexiao Zou, Liumeng Xue, Mingxuan Wang, Songting Liu, Tze Ying Tang, Xi Chen, Xueyao Zhang, Yicheng Gu, Yuancheng Wang, Zhizheng Wu, Zihao Fang","submitted_at":"2023-12-15T16:23:21Z","abstract_excerpt":"Amphion is an open-source toolkit for Audio, Music, and Speech Generation, targeting to ease the way for junior researchers and engineers into these fields. It presents a unified framework that includes diverse generation tasks and models, with the added bonus of being easily extendable for new incorporation. The toolkit is designed with beginner-friendly workflows and pre-trained models, allowing both beginners and seasoned researchers to kick-start their projects with relative ease. The initial release of Amphion v0.1 supports a range of tasks including Text to Speech (TTS), Text to Audio (T"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.09911","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.09911/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.09911","created_at":"2026-07-05T09:07:25.596928+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.09911v3","created_at":"2026-07-05T09:07:25.596928+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.09911","created_at":"2026-07-05T09:07:25.596928+00:00"},{"alias_kind":"pith_short_12","alias_value":"GLREUXZGGM4F","created_at":"2026-07-05T09:07:25.596928+00:00"},{"alias_kind":"pith_short_16","alias_value":"GLREUXZGGM4FZCWE","created_at":"2026-07-05T09:07:25.596928+00:00"},{"alias_kind":"pith_short_8","alias_value":"GLREUXZG","created_at":"2026-07-05T09:07:25.596928+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01238","citing_title":"SPARCLE: SPeaker-aware Aligned Representations via Contrastive Language Embeddings","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31247","citing_title":"FlexiSLM: A Dynamic and Controllable Frame Rate Spoken Language Model","ref_index":120,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GLREUXZGGM4FZCWEJLSEY53WBK","json":"https://pith.science/pith/GLREUXZGGM4FZCWEJLSEY53WBK.json","graph_json":"https://pith.science/api/pith-number/GLREUXZGGM4FZCWEJLSEY53WBK/graph.json","events_json":"https://pith.science/api/pith-number/GLREUXZGGM4FZCWEJLSEY53WBK/events.json","paper":"https://pith.science/paper/GLREUXZG"},"agent_actions":{"view_html":"https://pith.science/pith/GLREUXZGGM4FZCWEJLSEY53WBK","download_json":"https://pith.science/pith/GLREUXZGGM4FZCWEJLSEY53WBK.json","view_paper":"https://pith.science/paper/GLREUXZG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.09911&json=true","fetch_graph":"https://pith.science/api/pith-number/GLREUXZGGM4FZCWEJLSEY53WBK/graph.json","fetch_events":"https://pith.science/api/pith-number/GLREUXZGGM4FZCWEJLSEY53WBK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GLREUXZGGM4FZCWEJLSEY53WBK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GLREUXZGGM4FZCWEJLSEY53WBK/action/storage_attestation","attest_author":"https://pith.science/pith/GLREUXZGGM4FZCWEJLSEY53WBK/action/author_attestation","sign_citation":"https://pith.science/pith/GLREUXZGGM4FZCWEJLSEY53WBK/action/citation_signature","submit_replication":"https://pith.science/pith/GLREUXZGGM4FZCWEJLSEY53WBK/action/replication_record"}},"created_at":"2026-07-05T09:07:25.596928+00:00","updated_at":"2026-07-05T09:07:25.596928+00:00"}