{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:WS4GUIRVELX3QDIZZ4QHETF65X","short_pith_number":"pith:WS4GUIRV","schema_version":"1.0","canonical_sha256":"b4b86a223522efb80d19cf20724cbeedd1fee0e0b21b5756c377b866e1b602c7","source":{"kind":"arxiv","id":"2507.22746","version":2},"attestation_state":"computed","paper":{"title":"Next Tokens Denoising for Speech Synthesis","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","eess.AS"],"primary_cat":"cs.SD","authors_text":"Bohan Li, Chong Zhang, Gang Wang, Lei He, Ruiqing Xue, Sheng Zhao, Shujie Liu, Yanqing Liu, Yao Qian, Yufei Liu","submitted_at":"2025-07-30T15:03:36Z","abstract_excerpt":"While diffusion and autoregressive (AR) models have significantly advanced generative modeling, they each present distinct limitations. AR models, which rely on causal attention, cannot exploit future context and suffer from slow generation speeds. Conversely, diffusion models struggle with key-value (KV) caching. To overcome these challenges, we introduce Dragon-FM, a novel text-to-speech (TTS) design that unifies AR and flow-matching. This model processes 48 kHz audio codec tokens in chunks at a compact rate of 12.5 tokens per second. This design enables AR modeling across chunks, ensuring g"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.22746","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SD","submitted_at":"2025-07-30T15:03:36Z","cross_cats_sorted":["cs.CL","eess.AS"],"title_canon_sha256":"3d692dcce90663a051cc5b53babc457f81c41b818463562d5fa3f1986263dfc9","abstract_canon_sha256":"ba4cd40bc13ce498628c71c148c8267109a996b4be66e062fb1fba353e53827d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:46:46.468535Z","signature_b64":"gct8paIDL+cXH+0gS9oyabhGtWYTrEFSsJbUVkCZ9IGp9tXo7XAr3jQ2qxFzGaH++V0RmsETNoRfxjCbrCp+DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b4b86a223522efb80d19cf20724cbeedd1fee0e0b21b5756c377b866e1b602c7","last_reissued_at":"2026-07-05T11:46:46.468066Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:46:46.468066Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Next Tokens Denoising for Speech Synthesis","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","eess.AS"],"primary_cat":"cs.SD","authors_text":"Bohan Li, Chong Zhang, Gang Wang, Lei He, Ruiqing Xue, Sheng Zhao, Shujie Liu, Yanqing Liu, Yao Qian, Yufei Liu","submitted_at":"2025-07-30T15:03:36Z","abstract_excerpt":"While diffusion and autoregressive (AR) models have significantly advanced generative modeling, they each present distinct limitations. AR models, which rely on causal attention, cannot exploit future context and suffer from slow generation speeds. Conversely, diffusion models struggle with key-value (KV) caching. To overcome these challenges, we introduce Dragon-FM, a novel text-to-speech (TTS) design that unifies AR and flow-matching. This model processes 48 kHz audio codec tokens in chunks at a compact rate of 12.5 tokens per second. This design enables AR modeling across chunks, ensuring g"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.22746","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.22746/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.22746","created_at":"2026-07-05T11:46:46.468121+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.22746v2","created_at":"2026-07-05T11:46:46.468121+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.22746","created_at":"2026-07-05T11:46:46.468121+00:00"},{"alias_kind":"pith_short_12","alias_value":"WS4GUIRVELX3","created_at":"2026-07-05T11:46:46.468121+00:00"},{"alias_kind":"pith_short_16","alias_value":"WS4GUIRVELX3QDIZ","created_at":"2026-07-05T11:46:46.468121+00:00"},{"alias_kind":"pith_short_8","alias_value":"WS4GUIRV","created_at":"2026-07-05T11:46:46.468121+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WS4GUIRVELX3QDIZZ4QHETF65X","json":"https://pith.science/pith/WS4GUIRVELX3QDIZZ4QHETF65X.json","graph_json":"https://pith.science/api/pith-number/WS4GUIRVELX3QDIZZ4QHETF65X/graph.json","events_json":"https://pith.science/api/pith-number/WS4GUIRVELX3QDIZZ4QHETF65X/events.json","paper":"https://pith.science/paper/WS4GUIRV"},"agent_actions":{"view_html":"https://pith.science/pith/WS4GUIRVELX3QDIZZ4QHETF65X","download_json":"https://pith.science/pith/WS4GUIRVELX3QDIZZ4QHETF65X.json","view_paper":"https://pith.science/paper/WS4GUIRV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.22746&json=true","fetch_graph":"https://pith.science/api/pith-number/WS4GUIRVELX3QDIZZ4QHETF65X/graph.json","fetch_events":"https://pith.science/api/pith-number/WS4GUIRVELX3QDIZZ4QHETF65X/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WS4GUIRVELX3QDIZZ4QHETF65X/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WS4GUIRVELX3QDIZZ4QHETF65X/action/storage_attestation","attest_author":"https://pith.science/pith/WS4GUIRVELX3QDIZZ4QHETF65X/action/author_attestation","sign_citation":"https://pith.science/pith/WS4GUIRVELX3QDIZZ4QHETF65X/action/citation_signature","submit_replication":"https://pith.science/pith/WS4GUIRVELX3QDIZZ4QHETF65X/action/replication_record"}},"created_at":"2026-07-05T11:46:46.468121+00:00","updated_at":"2026-07-05T11:46:46.468121+00:00"}