{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:73VYDAKL5DFGUEHOWYQFHNJCHA","short_pith_number":"pith:73VYDAKL","schema_version":"1.0","canonical_sha256":"feeb81814be8ca6a10eeb62053b52238017ec17ca872df1aa991a4dd6db63c4d","source":{"kind":"arxiv","id":"2206.05408","version":3},"attestation_state":"computed","paper":{"title":"Multi-instrument Music Synthesis with Spectrogram Diffusion","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"Adam Roberts, Curtis Hawthorne, Ethan Manilow, Ian Simon, Jesse Engel, Josh Gardner, Neil Zeghidour","submitted_at":"2022-06-11T03:26:15Z","abstract_excerpt":"An ideal music synthesizer should be both interactive and expressive, generating high-fidelity audio in realtime for arbitrary combinations of instruments and notes. Recent neural synthesizers have exhibited a tradeoff between domain-specific models that offer detailed control of only specific instruments, or raw waveform models that can train on any music but with minimal control and slow generation. In this work, we focus on a middle ground of neural synthesizers that can generate audio from MIDI sequences with arbitrary combinations of instruments in realtime. This enables training on a wid"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2206.05408","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2022-06-11T03:26:15Z","cross_cats_sorted":["cs.LG","eess.AS"],"title_canon_sha256":"dd8b089bce3245ca0d49d5652435af096271e0edab119d9b1d1258ab6d819ad3","abstract_canon_sha256":"a08daffcd9f926bfacf0e4c3b1185e554a70757e9b008f4e321775e2d32b3097"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:24:38.582171Z","signature_b64":"Tb38e8zyE5comuT2Nf1UIxQm0SCF7Nhc3K7CII2Ckq0MNoQD01APsaF0F7DUaTArG+UpS7yFnpHDYuUykEDIBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"feeb81814be8ca6a10eeb62053b52238017ec17ca872df1aa991a4dd6db63c4d","last_reissued_at":"2026-07-05T05:24:38.581730Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:24:38.581730Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multi-instrument Music Synthesis with Spectrogram Diffusion","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"Adam Roberts, Curtis Hawthorne, Ethan Manilow, Ian Simon, Jesse Engel, Josh Gardner, Neil Zeghidour","submitted_at":"2022-06-11T03:26:15Z","abstract_excerpt":"An ideal music synthesizer should be both interactive and expressive, generating high-fidelity audio in realtime for arbitrary combinations of instruments and notes. Recent neural synthesizers have exhibited a tradeoff between domain-specific models that offer detailed control of only specific instruments, or raw waveform models that can train on any music but with minimal control and slow generation. In this work, we focus on a middle ground of neural synthesizers that can generate audio from MIDI sequences with arbitrary combinations of instruments in realtime. This enables training on a wid"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.05408","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2206.05408/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2206.05408","created_at":"2026-07-05T05:24:38.581791+00:00"},{"alias_kind":"arxiv_version","alias_value":"2206.05408v3","created_at":"2026-07-05T05:24:38.581791+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.05408","created_at":"2026-07-05T05:24:38.581791+00:00"},{"alias_kind":"pith_short_12","alias_value":"73VYDAKL5DFG","created_at":"2026-07-05T05:24:38.581791+00:00"},{"alias_kind":"pith_short_16","alias_value":"73VYDAKL5DFGUEHO","created_at":"2026-07-05T05:24:38.581791+00:00"},{"alias_kind":"pith_short_8","alias_value":"73VYDAKL","created_at":"2026-07-05T05:24:38.581791+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2301.11325","citing_title":"MusicLM: Generating Music From Text","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10281","citing_title":"Drum Synthesis from Expressive Drum Grids via Neural Audio Codecs","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/73VYDAKL5DFGUEHOWYQFHNJCHA","json":"https://pith.science/pith/73VYDAKL5DFGUEHOWYQFHNJCHA.json","graph_json":"https://pith.science/api/pith-number/73VYDAKL5DFGUEHOWYQFHNJCHA/graph.json","events_json":"https://pith.science/api/pith-number/73VYDAKL5DFGUEHOWYQFHNJCHA/events.json","paper":"https://pith.science/paper/73VYDAKL"},"agent_actions":{"view_html":"https://pith.science/pith/73VYDAKL5DFGUEHOWYQFHNJCHA","download_json":"https://pith.science/pith/73VYDAKL5DFGUEHOWYQFHNJCHA.json","view_paper":"https://pith.science/paper/73VYDAKL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2206.05408&json=true","fetch_graph":"https://pith.science/api/pith-number/73VYDAKL5DFGUEHOWYQFHNJCHA/graph.json","fetch_events":"https://pith.science/api/pith-number/73VYDAKL5DFGUEHOWYQFHNJCHA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/73VYDAKL5DFGUEHOWYQFHNJCHA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/73VYDAKL5DFGUEHOWYQFHNJCHA/action/storage_attestation","attest_author":"https://pith.science/pith/73VYDAKL5DFGUEHOWYQFHNJCHA/action/author_attestation","sign_citation":"https://pith.science/pith/73VYDAKL5DFGUEHOWYQFHNJCHA/action/citation_signature","submit_replication":"https://pith.science/pith/73VYDAKL5DFGUEHOWYQFHNJCHA/action/replication_record"}},"created_at":"2026-07-05T05:24:38.581791+00:00","updated_at":"2026-07-05T05:24:38.581791+00:00"}