{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ATQPWCAGD4IZGH7DLB6FU2M4J4","short_pith_number":"pith:ATQPWCAG","schema_version":"1.0","canonical_sha256":"04e0fb08061f11931fe3587c5a699c4f3b371a17a0f1c66016788c19020d96ab","source":{"kind":"arxiv","id":"2305.18474","version":1},"attestation_state":"computed","paper":{"title":"Make-An-Audio 2: Temporal-Enhanced Text-to-Audio Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.MM","eess.AS"],"primary_cat":"cs.SD","authors_text":"Chen Zhang, Dongchao Yang, Jiawei Huang, Jinglin Liu, Rongjie Huang, Xiang Yin, Yi Ren, Zejun Ma, Zhenhui Ye, Zhou Zhao","submitted_at":"2023-05-29T10:41:28Z","abstract_excerpt":"Large diffusion models have been successful in text-to-audio (T2A) synthesis tasks, but they often suffer from common issues such as semantic misalignment and poor temporal consistency due to limited natural language understanding and data scarcity. Additionally, 2D spatial structures widely used in T2A works lead to unsatisfactory audio quality when generating variable-length audio samples since they do not adequately prioritize temporal information. To address these challenges, we propose Make-an-Audio 2, a latent diffusion-based T2A method that builds on the success of Make-an-Audio. Our ap"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.18474","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2023-05-29T10:41:28Z","cross_cats_sorted":["cs.LG","cs.MM","eess.AS"],"title_canon_sha256":"4a74e950d900b5722a344bbdc9eba5049a87ae96ffa0afaf8bff48ff3df9cbcd","abstract_canon_sha256":"ac8c61e3fe50a9743653eb658c7d1feecbcd426c6269588e7b97647e2f94f396"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:15:01.259014Z","signature_b64":"19idt0Si4q1RxBkoev3Ai6shXSXIMIyB617o7+vAO4KM5kQlBHk2BPjnzLEsMx2VAxa5RT+7+LCF0p8WqxeoAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"04e0fb08061f11931fe3587c5a699c4f3b371a17a0f1c66016788c19020d96ab","last_reissued_at":"2026-07-05T06:15:01.258554Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:15:01.258554Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Make-An-Audio 2: Temporal-Enhanced Text-to-Audio Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.MM","eess.AS"],"primary_cat":"cs.SD","authors_text":"Chen Zhang, Dongchao Yang, Jiawei Huang, Jinglin Liu, Rongjie Huang, Xiang Yin, Yi Ren, Zejun Ma, Zhenhui Ye, Zhou Zhao","submitted_at":"2023-05-29T10:41:28Z","abstract_excerpt":"Large diffusion models have been successful in text-to-audio (T2A) synthesis tasks, but they often suffer from common issues such as semantic misalignment and poor temporal consistency due to limited natural language understanding and data scarcity. Additionally, 2D spatial structures widely used in T2A works lead to unsatisfactory audio quality when generating variable-length audio samples since they do not adequately prioritize temporal information. To address these challenges, we propose Make-an-Audio 2, a latent diffusion-based T2A method that builds on the success of Make-an-Audio. Our ap"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.18474","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.18474/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.18474","created_at":"2026-07-05T06:15:01.258607+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.18474v1","created_at":"2026-07-05T06:15:01.258607+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.18474","created_at":"2026-07-05T06:15:01.258607+00:00"},{"alias_kind":"pith_short_12","alias_value":"ATQPWCAGD4IZ","created_at":"2026-07-05T06:15:01.258607+00:00"},{"alias_kind":"pith_short_16","alias_value":"ATQPWCAGD4IZGH7D","created_at":"2026-07-05T06:15:01.258607+00:00"},{"alias_kind":"pith_short_8","alias_value":"ATQPWCAG","created_at":"2026-07-05T06:15:01.258607+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05196","citing_title":"Unified Audio Intelligence Without Regressing on Text Intelligence","ref_index":66,"is_internal_anchor":true},{"citing_arxiv_id":"2606.23064","citing_title":"STAR-VAE: Structured Topology-Aware Regularization for Audio Reconstruction and Generation","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20101","citing_title":"RFM-Editing 2: Text-Guided Audio Editing with Rectified Flow Matching and Coarse-to-Fine Diffusion Transformers","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20101","citing_title":"RFM-Editing 2: Text-Guided Audio Editing with Rectified Flow Matching and Coarse-to-Fine Diffusion Transformers","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12555","citing_title":"AudioX-Turbo: A Unified Framework for Efficient Anything-to-Audio Generation","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31530","citing_title":"UNISON: A Unified Sound Generation and Editing Framework via Deep LLM Fusion","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2509.06027","citing_title":"DreamAudio: Customized Text-to-Audio Generation with Diffusion Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2509.14003","citing_title":"RFM-Editing: Rectified Flow Matching for Text-guided Audio Editing","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2509.23727","citing_title":"AudioMoG: Guiding Audio Generation with Mixture-of-Guidance","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2510.01284","citing_title":"Ovi: Twin Backbone Cross-Modal Fusion for Audio-Video Generation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2601.02731","citing_title":"Omni2Sound: Towards Unified Video-Text-to-Audio Generation","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00329","citing_title":"Fast Text-to-Audio Generation with One-Step Sampling via Energy-Scoring and Auxiliary Contextual Representation Distillation","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05731","citing_title":"FoleyDesigner: Immersive Stereo Foley Generation with Precise Spatio-Temporal Alignment for Film Clips","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14707","citing_title":"Geo2Sound: A Scalable Geo-Aligned Framework for Soundscape Generation from Satellite Imagery","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ATQPWCAGD4IZGH7DLB6FU2M4J4","json":"https://pith.science/pith/ATQPWCAGD4IZGH7DLB6FU2M4J4.json","graph_json":"https://pith.science/api/pith-number/ATQPWCAGD4IZGH7DLB6FU2M4J4/graph.json","events_json":"https://pith.science/api/pith-number/ATQPWCAGD4IZGH7DLB6FU2M4J4/events.json","paper":"https://pith.science/paper/ATQPWCAG"},"agent_actions":{"view_html":"https://pith.science/pith/ATQPWCAGD4IZGH7DLB6FU2M4J4","download_json":"https://pith.science/pith/ATQPWCAGD4IZGH7DLB6FU2M4J4.json","view_paper":"https://pith.science/paper/ATQPWCAG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.18474&json=true","fetch_graph":"https://pith.science/api/pith-number/ATQPWCAGD4IZGH7DLB6FU2M4J4/graph.json","fetch_events":"https://pith.science/api/pith-number/ATQPWCAGD4IZGH7DLB6FU2M4J4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ATQPWCAGD4IZGH7DLB6FU2M4J4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ATQPWCAGD4IZGH7DLB6FU2M4J4/action/storage_attestation","attest_author":"https://pith.science/pith/ATQPWCAGD4IZGH7DLB6FU2M4J4/action/author_attestation","sign_citation":"https://pith.science/pith/ATQPWCAGD4IZGH7DLB6FU2M4J4/action/citation_signature","submit_replication":"https://pith.science/pith/ATQPWCAGD4IZGH7DLB6FU2M4J4/action/replication_record"}},"created_at":"2026-07-05T06:15:01.258607+00:00","updated_at":"2026-07-05T06:15:01.258607+00:00"}