{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MPG3ST4BUQBABZAHTMS23H3OXZ","short_pith_number":"pith:MPG3ST4B","schema_version":"1.0","canonical_sha256":"63cdb94f81a40200e4079b25ad9f6ebe70e0ab0819cbcdd03080ea5a38cec57e","source":{"kind":"arxiv","id":"2406.05298","version":2},"attestation_state":"computed","paper":{"title":"Spectral Codecs: Improving Non-Autoregressive Speech Synthesis with Spectrogram-Based Audio Codecs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"eess.AS","authors_text":"Ante Juki\\'c, Jason Li, Kunal Dhawan, Nithin Rao Koluguri, Ryan Langman","submitted_at":"2024-06-07T23:47:51Z","abstract_excerpt":"Historically, most speech models in machine-learning have used the mel-spectrogram as a speech representation. Recently, discrete audio tokens produced by neural audio codecs have become a popular alternate speech representation for speech synthesis tasks such as text-to-speech (TTS). However, the data distribution produced by such codecs is too complex for some TTS models to predict, typically requiring large autoregressive models to get good quality. Most existing audio codecs use Residual Vector Quantization (RVQ) to compress and reconstruct the time-domain audio signal. We propose a spectr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.05298","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2024-06-07T23:47:51Z","cross_cats_sorted":[],"title_canon_sha256":"6b8e85434040ac1d10d11eec52ac53eb703d128780c64a7d38d0440f05e3f681","abstract_canon_sha256":"71a8a18bc500025a5ed86f0560555585084168c4047ceecf4f3467ac4d079871"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:15:37.756105Z","signature_b64":"R1MvzX1Cjw/wS/SgsF6Qr4QMSsCIXAaCvgV/MX9aUoHjm0gXS+cdnuuVHJKHJc9o726VDox6xX6jfWNlfN+6Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"63cdb94f81a40200e4079b25ad9f6ebe70e0ab0819cbcdd03080ea5a38cec57e","last_reissued_at":"2026-07-05T11:15:37.755583Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:15:37.755583Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Spectral Codecs: Improving Non-Autoregressive Speech Synthesis with Spectrogram-Based Audio Codecs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"eess.AS","authors_text":"Ante Juki\\'c, Jason Li, Kunal Dhawan, Nithin Rao Koluguri, Ryan Langman","submitted_at":"2024-06-07T23:47:51Z","abstract_excerpt":"Historically, most speech models in machine-learning have used the mel-spectrogram as a speech representation. Recently, discrete audio tokens produced by neural audio codecs have become a popular alternate speech representation for speech synthesis tasks such as text-to-speech (TTS). However, the data distribution produced by such codecs is too complex for some TTS models to predict, typically requiring large autoregressive models to get good quality. Most existing audio codecs use Residual Vector Quantization (RVQ) to compress and reconstruct the time-domain audio signal. We propose a spectr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.05298","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.05298/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.05298","created_at":"2026-07-05T11:15:37.755649+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.05298v2","created_at":"2026-07-05T11:15:37.755649+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.05298","created_at":"2026-07-05T11:15:37.755649+00:00"},{"alias_kind":"pith_short_12","alias_value":"MPG3ST4BUQBA","created_at":"2026-07-05T11:15:37.755649+00:00"},{"alias_kind":"pith_short_16","alias_value":"MPG3ST4BUQBABZAH","created_at":"2026-07-05T11:15:37.755649+00:00"},{"alias_kind":"pith_short_8","alias_value":"MPG3ST4B","created_at":"2026-07-05T11:15:37.755649+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11631","citing_title":"Benchmarking Neural Speech Compression from a Rate-Distortion Perspective","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25669","citing_title":"Ultra-Low-Bitrate Mel-Spectrogram-based Neural Speech Coding with Flow-Matching-based Refinement and Vocoding-driven Reconstruction","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15831","citing_title":"Modeling Music as a Time-Frequency Image: A 2D Tokenizer for Music Generation","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MPG3ST4BUQBABZAHTMS23H3OXZ","json":"https://pith.science/pith/MPG3ST4BUQBABZAHTMS23H3OXZ.json","graph_json":"https://pith.science/api/pith-number/MPG3ST4BUQBABZAHTMS23H3OXZ/graph.json","events_json":"https://pith.science/api/pith-number/MPG3ST4BUQBABZAHTMS23H3OXZ/events.json","paper":"https://pith.science/paper/MPG3ST4B"},"agent_actions":{"view_html":"https://pith.science/pith/MPG3ST4BUQBABZAHTMS23H3OXZ","download_json":"https://pith.science/pith/MPG3ST4BUQBABZAHTMS23H3OXZ.json","view_paper":"https://pith.science/paper/MPG3ST4B","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.05298&json=true","fetch_graph":"https://pith.science/api/pith-number/MPG3ST4BUQBABZAHTMS23H3OXZ/graph.json","fetch_events":"https://pith.science/api/pith-number/MPG3ST4BUQBABZAHTMS23H3OXZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MPG3ST4BUQBABZAHTMS23H3OXZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MPG3ST4BUQBABZAHTMS23H3OXZ/action/storage_attestation","attest_author":"https://pith.science/pith/MPG3ST4BUQBABZAHTMS23H3OXZ/action/author_attestation","sign_citation":"https://pith.science/pith/MPG3ST4BUQBABZAHTMS23H3OXZ/action/citation_signature","submit_replication":"https://pith.science/pith/MPG3ST4BUQBABZAHTMS23H3OXZ/action/replication_record"}},"created_at":"2026-07-05T11:15:37.755649+00:00","updated_at":"2026-07-05T11:15:37.755649+00:00"}