{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:CCQNRIXXY6AGNAGYZIXEOUK5W2","short_pith_number":"pith:CCQNRIXX","schema_version":"1.0","canonical_sha256":"10a0d8a2f7c7806680d8ca2e47515db681161f07db0830aac263cc7d77eca5a6","source":{"kind":"arxiv","id":"2311.10057","version":3},"attestation_state":"computed","paper":{"title":"The Song Describer Dataset: a Corpus of Audio Captions for Music-and-Language Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","eess.AS"],"primary_cat":"cs.SD","authors_text":"Benno Weck, Dmitry Bogdanov, Elio Quinton, Emmanouil Benetos, Gy\\\"orgy Fazekas, Ilaria Manco, Juhan Nam, Ke Chen, Minz Won, Philip Tovstogan, Seungheon Doh, Yixiao Zhang, Yusong Wu","submitted_at":"2023-11-16T17:52:21Z","abstract_excerpt":"We introduce the Song Describer dataset (SDD), a new crowdsourced corpus of high-quality audio-caption pairs, designed for the evaluation of music-and-language models. The dataset consists of 1.1k human-written natural language descriptions of 706 music recordings, all publicly accessible and released under Creative Common licenses. To showcase the use of our dataset, we benchmark popular models on three key music-and-language tasks (music captioning, text-to-music generation and music-language retrieval). Our experiments highlight the importance of cross-dataset evaluation and offer insights "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.10057","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2023-11-16T17:52:21Z","cross_cats_sorted":["cs.AI","cs.CL","eess.AS"],"title_canon_sha256":"59407bc11bf780791e71cea4d839eadaaefc9d20cacb70d642f63a581a35e487","abstract_canon_sha256":"3421b6ba0e64cd7b8903c38c41dac555d8847318abe405104af825cb8684edac"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:16:03.617354Z","signature_b64":"W+e1QpQAtXlABSEffQbMZxMRYomO7zTu1+06k13RBL9grVGWq/8E1ttH3nF5t9XzabALjPMVkgr3Rnna4sY7Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"10a0d8a2f7c7806680d8ca2e47515db681161f07db0830aac263cc7d77eca5a6","last_reissued_at":"2026-07-05T07:16:03.616777Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:16:03.616777Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Song Describer Dataset: a Corpus of Audio Captions for Music-and-Language Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","eess.AS"],"primary_cat":"cs.SD","authors_text":"Benno Weck, Dmitry Bogdanov, Elio Quinton, Emmanouil Benetos, Gy\\\"orgy Fazekas, Ilaria Manco, Juhan Nam, Ke Chen, Minz Won, Philip Tovstogan, Seungheon Doh, Yixiao Zhang, Yusong Wu","submitted_at":"2023-11-16T17:52:21Z","abstract_excerpt":"We introduce the Song Describer dataset (SDD), a new crowdsourced corpus of high-quality audio-caption pairs, designed for the evaluation of music-and-language models. The dataset consists of 1.1k human-written natural language descriptions of 706 music recordings, all publicly accessible and released under Creative Common licenses. To showcase the use of our dataset, we benchmark popular models on three key music-and-language tasks (music captioning, text-to-music generation and music-language retrieval). Our experiments highlight the importance of cross-dataset evaluation and offer insights "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.10057","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.10057/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.10057","created_at":"2026-07-05T07:16:03.616849+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.10057v3","created_at":"2026-07-05T07:16:03.616849+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.10057","created_at":"2026-07-05T07:16:03.616849+00:00"},{"alias_kind":"pith_short_12","alias_value":"CCQNRIXXY6AG","created_at":"2026-07-05T07:16:03.616849+00:00"},{"alias_kind":"pith_short_16","alias_value":"CCQNRIXXY6AGNAGY","created_at":"2026-07-05T07:16:03.616849+00:00"},{"alias_kind":"pith_short_8","alias_value":"CCQNRIXX","created_at":"2026-07-05T07:16:03.616849+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24307","citing_title":"Real-Time Interactive Music Generation via Data-Free Streaming Consistency Distillation","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23064","citing_title":"STAR-VAE: Structured Topology-Aware Regularization for Audio Reconstruction and Generation","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22717","citing_title":"Live Music Diffusion Models: Efficient Fine-Tuning and Post-Training of Interactive Diffusion Music Generators","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16681","citing_title":"A Survey of Advancing Audio Super-Resolution and Bandwidth Extension from Discriminative to Generative Models","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15831","citing_title":"Modeling Music as a Time-Frequency Image: A 2D Tokenizer for Music Generation","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2507.23511","citing_title":"MECAT: A Multi-Experts Constructed Benchmark for Fine-Grained Audio Understanding Tasks","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2601.02954","citing_title":"The World is Not Mono: Enabling Spatial Understanding in Large Audio-Language Models","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CCQNRIXXY6AGNAGYZIXEOUK5W2","json":"https://pith.science/pith/CCQNRIXXY6AGNAGYZIXEOUK5W2.json","graph_json":"https://pith.science/api/pith-number/CCQNRIXXY6AGNAGYZIXEOUK5W2/graph.json","events_json":"https://pith.science/api/pith-number/CCQNRIXXY6AGNAGYZIXEOUK5W2/events.json","paper":"https://pith.science/paper/CCQNRIXX"},"agent_actions":{"view_html":"https://pith.science/pith/CCQNRIXXY6AGNAGYZIXEOUK5W2","download_json":"https://pith.science/pith/CCQNRIXXY6AGNAGYZIXEOUK5W2.json","view_paper":"https://pith.science/paper/CCQNRIXX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.10057&json=true","fetch_graph":"https://pith.science/api/pith-number/CCQNRIXXY6AGNAGYZIXEOUK5W2/graph.json","fetch_events":"https://pith.science/api/pith-number/CCQNRIXXY6AGNAGYZIXEOUK5W2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CCQNRIXXY6AGNAGYZIXEOUK5W2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CCQNRIXXY6AGNAGYZIXEOUK5W2/action/storage_attestation","attest_author":"https://pith.science/pith/CCQNRIXXY6AGNAGYZIXEOUK5W2/action/author_attestation","sign_citation":"https://pith.science/pith/CCQNRIXXY6AGNAGYZIXEOUK5W2/action/citation_signature","submit_replication":"https://pith.science/pith/CCQNRIXXY6AGNAGYZIXEOUK5W2/action/replication_record"}},"created_at":"2026-07-05T07:16:03.616849+00:00","updated_at":"2026-07-05T07:16:03.616849+00:00"}