{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BPHOJEMZ5LADP3Y54FL5SLRPMG","short_pith_number":"pith:BPHOJEMZ","schema_version":"1.0","canonical_sha256":"0bcee49199eac037ef1de157d92e2f61a05d39c88504a43602bb54d4ccc1ad7b","source":{"kind":"arxiv","id":"2407.02543","version":2},"attestation_state":"computed","paper":{"title":"Towards the Next Frontier in Speech Representation Learning Using Disentanglement","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Sriram Ganapathy, Varun Krishna","submitted_at":"2024-07-02T07:13:35Z","abstract_excerpt":"The popular frameworks for self-supervised learning of speech representations have largely focused on frame-level masked prediction of speech regions. While this has shown promising downstream task performance for speech recognition and related tasks, this has largely ignored factors of speech that are encoded at coarser level, like characteristics of the speaker or channel that remain consistent through-out a speech utterance. In this work, we propose a framework for Learning Disentangled Self Supervised (termed as Learn2Diss) representations of speech, which consists of frame-level and an ut"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.02543","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-02T07:13:35Z","cross_cats_sorted":["cs.AI","cs.LG","cs.SD","eess.AS"],"title_canon_sha256":"b73139c8ef81fe213e6acb63e27dd4077f8d1b05c3980b5663b30dbd766571d7","abstract_canon_sha256":"f9596f8019e853da540e71b3e9fc428d860a234b9062b8f8ae07d94c877b477e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:39:41.497209Z","signature_b64":"UI5T/W6MtUCA5e4eAHz1+QNAjv5MdSRLEIsUQnVArzHArXaxz8DUJrfCV8o/OEq9tuFOweUma9oyEmEydKoKDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0bcee49199eac037ef1de157d92e2f61a05d39c88504a43602bb54d4ccc1ad7b","last_reissued_at":"2026-07-05T11:39:41.496694Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:39:41.496694Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards the Next Frontier in Speech Representation Learning Using Disentanglement","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Sriram Ganapathy, Varun Krishna","submitted_at":"2024-07-02T07:13:35Z","abstract_excerpt":"The popular frameworks for self-supervised learning of speech representations have largely focused on frame-level masked prediction of speech regions. While this has shown promising downstream task performance for speech recognition and related tasks, this has largely ignored factors of speech that are encoded at coarser level, like characteristics of the speaker or channel that remain consistent through-out a speech utterance. In this work, we propose a framework for Learning Disentangled Self Supervised (termed as Learn2Diss) representations of speech, which consists of frame-level and an ut"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.02543","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.02543/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.02543","created_at":"2026-07-05T11:39:41.496756+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.02543v2","created_at":"2026-07-05T11:39:41.496756+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.02543","created_at":"2026-07-05T11:39:41.496756+00:00"},{"alias_kind":"pith_short_12","alias_value":"BPHOJEMZ5LAD","created_at":"2026-07-05T11:39:41.496756+00:00"},{"alias_kind":"pith_short_16","alias_value":"BPHOJEMZ5LADP3Y5","created_at":"2026-07-05T11:39:41.496756+00:00"},{"alias_kind":"pith_short_8","alias_value":"BPHOJEMZ","created_at":"2026-07-05T11:39:41.496756+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.16532","citing_title":"Dual-Granularity Orthogonal Disentanglement for Generalizable Audio Deepfake Detection","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BPHOJEMZ5LADP3Y54FL5SLRPMG","json":"https://pith.science/pith/BPHOJEMZ5LADP3Y54FL5SLRPMG.json","graph_json":"https://pith.science/api/pith-number/BPHOJEMZ5LADP3Y54FL5SLRPMG/graph.json","events_json":"https://pith.science/api/pith-number/BPHOJEMZ5LADP3Y54FL5SLRPMG/events.json","paper":"https://pith.science/paper/BPHOJEMZ"},"agent_actions":{"view_html":"https://pith.science/pith/BPHOJEMZ5LADP3Y54FL5SLRPMG","download_json":"https://pith.science/pith/BPHOJEMZ5LADP3Y54FL5SLRPMG.json","view_paper":"https://pith.science/paper/BPHOJEMZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.02543&json=true","fetch_graph":"https://pith.science/api/pith-number/BPHOJEMZ5LADP3Y54FL5SLRPMG/graph.json","fetch_events":"https://pith.science/api/pith-number/BPHOJEMZ5LADP3Y54FL5SLRPMG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BPHOJEMZ5LADP3Y54FL5SLRPMG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BPHOJEMZ5LADP3Y54FL5SLRPMG/action/storage_attestation","attest_author":"https://pith.science/pith/BPHOJEMZ5LADP3Y54FL5SLRPMG/action/author_attestation","sign_citation":"https://pith.science/pith/BPHOJEMZ5LADP3Y54FL5SLRPMG/action/citation_signature","submit_replication":"https://pith.science/pith/BPHOJEMZ5LADP3Y54FL5SLRPMG/action/replication_record"}},"created_at":"2026-07-05T11:39:41.496756+00:00","updated_at":"2026-07-05T11:39:41.496756+00:00"}