{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:U2OWL7YTIWWQ3HSN5657J35RIV","short_pith_number":"pith:U2OWL7YT","schema_version":"1.0","canonical_sha256":"a69d65ff1345ad0d9e4defbbf4efb14547f929af9d366b27acc39b88968ffb5a","source":{"kind":"arxiv","id":"2010.11459","version":2},"attestation_state":"computed","paper":{"title":"A Framework for Generative and Contrastive Learning of Audio Representations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"Julius Smith, Prateek Verma","submitted_at":"2020-10-22T05:52:32Z","abstract_excerpt":"In this paper, we present a framework for contrastive learning for audio representations, in a self supervised frame work without access to any ground truth labels. The core idea in self supervised contrastive learning is to map an audio signal and its various augmented versions (representative of salient aspects of audio like pitch, timbre etc.) to a space where they are close together, and are separated from other different signals. In addition we also explore generative models based on state of the art transformer based architectures for learning latent spaces for audio signals, without acc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2010.11459","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2020-10-22T05:52:32Z","cross_cats_sorted":["cs.LG","eess.AS"],"title_canon_sha256":"9d813b3b93ce2b290b9100ee516b299ea0a75cdd77862dc084bebb03fab70293","abstract_canon_sha256":"4e2f35ca85667290fb49f75fa863e901359f32cfa346c27cc8d50114750d4e0c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:23:59.166191Z","signature_b64":"ysxHNiBreSlUp7fqiPziernLBC/RJPHWXZaYQHDN3mgdJ/nkKwHN6RUiH3R0Dar4hHgcMW4WsUZTRi5kNF9NBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a69d65ff1345ad0d9e4defbbf4efb14547f929af9d366b27acc39b88968ffb5a","last_reissued_at":"2026-07-05T02:23:59.165649Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:23:59.165649Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Framework for Generative and Contrastive Learning of Audio Representations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"Julius Smith, Prateek Verma","submitted_at":"2020-10-22T05:52:32Z","abstract_excerpt":"In this paper, we present a framework for contrastive learning for audio representations, in a self supervised frame work without access to any ground truth labels. The core idea in self supervised contrastive learning is to map an audio signal and its various augmented versions (representative of salient aspects of audio like pitch, timbre etc.) to a space where they are close together, and are separated from other different signals. In addition we also explore generative models based on state of the art transformer based architectures for learning latent spaces for audio signals, without acc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2010.11459","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2010.11459/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2010.11459","created_at":"2026-07-05T02:23:59.165710+00:00"},{"alias_kind":"arxiv_version","alias_value":"2010.11459v2","created_at":"2026-07-05T02:23:59.165710+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2010.11459","created_at":"2026-07-05T02:23:59.165710+00:00"},{"alias_kind":"pith_short_12","alias_value":"U2OWL7YTIWWQ","created_at":"2026-07-05T02:23:59.165710+00:00"},{"alias_kind":"pith_short_16","alias_value":"U2OWL7YTIWWQ3HSN","created_at":"2026-07-05T02:23:59.165710+00:00"},{"alias_kind":"pith_short_8","alias_value":"U2OWL7YT","created_at":"2026-07-05T02:23:59.165710+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.11449","citing_title":"Whisper-GPT -- Continuous Discrete Hybrid Representation Language Models For Speech And Music","ref_index":15,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/U2OWL7YTIWWQ3HSN5657J35RIV","json":"https://pith.science/pith/U2OWL7YTIWWQ3HSN5657J35RIV.json","graph_json":"https://pith.science/api/pith-number/U2OWL7YTIWWQ3HSN5657J35RIV/graph.json","events_json":"https://pith.science/api/pith-number/U2OWL7YTIWWQ3HSN5657J35RIV/events.json","paper":"https://pith.science/paper/U2OWL7YT"},"agent_actions":{"view_html":"https://pith.science/pith/U2OWL7YTIWWQ3HSN5657J35RIV","download_json":"https://pith.science/pith/U2OWL7YTIWWQ3HSN5657J35RIV.json","view_paper":"https://pith.science/paper/U2OWL7YT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2010.11459&json=true","fetch_graph":"https://pith.science/api/pith-number/U2OWL7YTIWWQ3HSN5657J35RIV/graph.json","fetch_events":"https://pith.science/api/pith-number/U2OWL7YTIWWQ3HSN5657J35RIV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/U2OWL7YTIWWQ3HSN5657J35RIV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/U2OWL7YTIWWQ3HSN5657J35RIV/action/storage_attestation","attest_author":"https://pith.science/pith/U2OWL7YTIWWQ3HSN5657J35RIV/action/author_attestation","sign_citation":"https://pith.science/pith/U2OWL7YTIWWQ3HSN5657J35RIV/action/citation_signature","submit_replication":"https://pith.science/pith/U2OWL7YTIWWQ3HSN5657J35RIV/action/replication_record"}},"created_at":"2026-07-05T02:23:59.165710+00:00","updated_at":"2026-07-05T02:23:59.165710+00:00"}