{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:NYGTGKWJUFP65BDQY6YHKTY5W2","short_pith_number":"pith:NYGTGKWJ","schema_version":"1.0","canonical_sha256":"6e0d332ac9a15fee8470c7b0754f1db686a0ce56b5f6bcd137313832d54b1df4","source":{"kind":"arxiv","id":"2309.05767","version":2},"attestation_state":"computed","paper":{"title":"Natural Language Supervision for General-Purpose Audio Representations","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.SD","authors_text":"Benjamin Elizalde, Huaming Wang, Soham Deshmukh","submitted_at":"2023-09-11T18:50:21Z","abstract_excerpt":"Audio-Language models jointly learn multimodal text and audio representations that enable Zero-Shot inference. Models rely on the encoders to create powerful representations of the input and generalize to multiple tasks ranging from sounds, music, and speech. Although models have achieved remarkable performance, there is still a performance gap with task-specific models. In this paper, we propose a Contrastive Language-Audio Pretraining model that is pretrained with a diverse collection of 4.6M audio-text pairs employing two innovative encoders for Zero-Shot inference. To learn audio represent"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.05767","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SD","submitted_at":"2023-09-11T18:50:21Z","cross_cats_sorted":["eess.AS"],"title_canon_sha256":"ee7d57de3024bc1cecd0c9b76ac708393c6a75576eacd9f12408bc43c4eeef51","abstract_canon_sha256":"ca298ebc5ad1f3447a9e1f4f866dc63f669548ba233306e1132c3b0c2d44537f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:42:19.878337Z","signature_b64":"Fes+hPAlBzEkCekyMKPMz1KTy7fpOARLNIlKgrvXBpcFvzSp6vqSd4Ry4VNwFiMDu182Em3qoGLjkJm+8oAeAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6e0d332ac9a15fee8470c7b0754f1db686a0ce56b5f6bcd137313832d54b1df4","last_reissued_at":"2026-07-05T07:42:19.877846Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:42:19.877846Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Natural Language Supervision for General-Purpose Audio Representations","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.SD","authors_text":"Benjamin Elizalde, Huaming Wang, Soham Deshmukh","submitted_at":"2023-09-11T18:50:21Z","abstract_excerpt":"Audio-Language models jointly learn multimodal text and audio representations that enable Zero-Shot inference. Models rely on the encoders to create powerful representations of the input and generalize to multiple tasks ranging from sounds, music, and speech. Although models have achieved remarkable performance, there is still a performance gap with task-specific models. In this paper, we propose a Contrastive Language-Audio Pretraining model that is pretrained with a diverse collection of 4.6M audio-text pairs employing two innovative encoders for Zero-Shot inference. To learn audio represent"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.05767","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.05767/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.05767","created_at":"2026-07-05T07:42:19.877903+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.05767v2","created_at":"2026-07-05T07:42:19.877903+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.05767","created_at":"2026-07-05T07:42:19.877903+00:00"},{"alias_kind":"pith_short_12","alias_value":"NYGTGKWJUFP6","created_at":"2026-07-05T07:42:19.877903+00:00"},{"alias_kind":"pith_short_16","alias_value":"NYGTGKWJUFP65BDQ","created_at":"2026-07-05T07:42:19.877903+00:00"},{"alias_kind":"pith_short_8","alias_value":"NYGTGKWJ","created_at":"2026-07-05T07:42:19.877903+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06615","citing_title":"FIGMA: Towards FIne-Grained Music retrievAl","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20847","citing_title":"Revisiting Content-Based Music Recommendation: Efficient Feature Aggregation from Large-Scale Music Models","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NYGTGKWJUFP65BDQY6YHKTY5W2","json":"https://pith.science/pith/NYGTGKWJUFP65BDQY6YHKTY5W2.json","graph_json":"https://pith.science/api/pith-number/NYGTGKWJUFP65BDQY6YHKTY5W2/graph.json","events_json":"https://pith.science/api/pith-number/NYGTGKWJUFP65BDQY6YHKTY5W2/events.json","paper":"https://pith.science/paper/NYGTGKWJ"},"agent_actions":{"view_html":"https://pith.science/pith/NYGTGKWJUFP65BDQY6YHKTY5W2","download_json":"https://pith.science/pith/NYGTGKWJUFP65BDQY6YHKTY5W2.json","view_paper":"https://pith.science/paper/NYGTGKWJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.05767&json=true","fetch_graph":"https://pith.science/api/pith-number/NYGTGKWJUFP65BDQY6YHKTY5W2/graph.json","fetch_events":"https://pith.science/api/pith-number/NYGTGKWJUFP65BDQY6YHKTY5W2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NYGTGKWJUFP65BDQY6YHKTY5W2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NYGTGKWJUFP65BDQY6YHKTY5W2/action/storage_attestation","attest_author":"https://pith.science/pith/NYGTGKWJUFP65BDQY6YHKTY5W2/action/author_attestation","sign_citation":"https://pith.science/pith/NYGTGKWJUFP65BDQY6YHKTY5W2/action/citation_signature","submit_replication":"https://pith.science/pith/NYGTGKWJUFP65BDQY6YHKTY5W2/action/replication_record"}},"created_at":"2026-07-05T07:42:19.877903+00:00","updated_at":"2026-07-05T07:42:19.877903+00:00"}