{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:AYGY35X3R3VFKO5AQNTPC4HNF4","short_pith_number":"pith:AYGY35X3","schema_version":"1.0","canonical_sha256":"060d8df6fb8eea553ba08366f170ed2f29369cfa6830f1c41bb1e734afa7f382","source":{"kind":"arxiv","id":"2402.04825","version":3},"attestation_state":"computed","paper":{"title":"Fast Timing-Conditioned Latent Audio Diffusion","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"CJ Carr, Jordi Pons, Josiah Taylor, Scott H. Hawley, Zach Evans","submitted_at":"2024-02-07T13:23:25Z","abstract_excerpt":"Generating long-form 44.1kHz stereo audio from text prompts can be computationally demanding. Further, most previous works do not tackle that music and sound effects naturally vary in their duration. Our research focuses on the efficient generation of long-form, variable-length stereo music and sounds at 44.1kHz using text prompts with a generative model. Stable Audio is based on latent diffusion, with its latent defined by a fully-convolutional variational autoencoder. It is conditioned on text prompts as well as timing embeddings, allowing for fine control over both the content and length of"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.04825","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2024-02-07T13:23:25Z","cross_cats_sorted":["cs.LG","eess.AS"],"title_canon_sha256":"363a080980ae51dbb27a04f1962618eb20205da5a7c69976b8374b6aa622d990","abstract_canon_sha256":"db6654becb7fffbe8f5f5818fb038679a23b43e16cd1bd4bf9b33ee29228771f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:18:33.220934Z","signature_b64":"dvC31VK5UI6aplhOmsfXHwVDyZx38NPb0ZjYazQCensG4zToVeJGc/Em7IUVN/K5s6daSRNZDllXLhGMT6XADg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"060d8df6fb8eea553ba08366f170ed2f29369cfa6830f1c41bb1e734afa7f382","last_reissued_at":"2026-07-05T08:18:33.220464Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:18:33.220464Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Fast Timing-Conditioned Latent Audio Diffusion","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"CJ Carr, Jordi Pons, Josiah Taylor, Scott H. Hawley, Zach Evans","submitted_at":"2024-02-07T13:23:25Z","abstract_excerpt":"Generating long-form 44.1kHz stereo audio from text prompts can be computationally demanding. Further, most previous works do not tackle that music and sound effects naturally vary in their duration. Our research focuses on the efficient generation of long-form, variable-length stereo music and sounds at 44.1kHz using text prompts with a generative model. Stable Audio is based on latent diffusion, with its latent defined by a fully-convolutional variational autoencoder. It is conditioned on text prompts as well as timing embeddings, allowing for fine control over both the content and length of"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.04825","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.04825/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.04825","created_at":"2026-07-05T08:18:33.220519+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.04825v3","created_at":"2026-07-05T08:18:33.220519+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.04825","created_at":"2026-07-05T08:18:33.220519+00:00"},{"alias_kind":"pith_short_12","alias_value":"AYGY35X3R3VF","created_at":"2026-07-05T08:18:33.220519+00:00"},{"alias_kind":"pith_short_16","alias_value":"AYGY35X3R3VFKO5A","created_at":"2026-07-05T08:18:33.220519+00:00"},{"alias_kind":"pith_short_8","alias_value":"AYGY35X3","created_at":"2026-07-05T08:18:33.220519+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10046","citing_title":"Inside the Latent Flow: Causal Deciphering of Attention Dynamics in Audio Separation Foundation Models","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07207","citing_title":"Entropy as a Structural Prior: How a Log-Barrier on DiT Belief Space Drives Musical Diversity and Development","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23275","citing_title":"Diffusion Domain Expansion: Learning to Coordinate Pre-trained Diffusion Models","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2601.02731","citing_title":"Omni2Sound: Towards Unified Video-Text-to-Audio Generation","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.01929","citing_title":"Woosh: A Sound Effects Foundation Model","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AYGY35X3R3VFKO5AQNTPC4HNF4","json":"https://pith.science/pith/AYGY35X3R3VFKO5AQNTPC4HNF4.json","graph_json":"https://pith.science/api/pith-number/AYGY35X3R3VFKO5AQNTPC4HNF4/graph.json","events_json":"https://pith.science/api/pith-number/AYGY35X3R3VFKO5AQNTPC4HNF4/events.json","paper":"https://pith.science/paper/AYGY35X3"},"agent_actions":{"view_html":"https://pith.science/pith/AYGY35X3R3VFKO5AQNTPC4HNF4","download_json":"https://pith.science/pith/AYGY35X3R3VFKO5AQNTPC4HNF4.json","view_paper":"https://pith.science/paper/AYGY35X3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.04825&json=true","fetch_graph":"https://pith.science/api/pith-number/AYGY35X3R3VFKO5AQNTPC4HNF4/graph.json","fetch_events":"https://pith.science/api/pith-number/AYGY35X3R3VFKO5AQNTPC4HNF4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AYGY35X3R3VFKO5AQNTPC4HNF4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AYGY35X3R3VFKO5AQNTPC4HNF4/action/storage_attestation","attest_author":"https://pith.science/pith/AYGY35X3R3VFKO5AQNTPC4HNF4/action/author_attestation","sign_citation":"https://pith.science/pith/AYGY35X3R3VFKO5AQNTPC4HNF4/action/citation_signature","submit_replication":"https://pith.science/pith/AYGY35X3R3VFKO5AQNTPC4HNF4/action/replication_record"}},"created_at":"2026-07-05T08:18:33.220519+00:00","updated_at":"2026-07-05T08:18:33.220519+00:00"}