{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:WETM7BN7SY7ARKROEMEKPNV2JG","short_pith_number":"pith:WETM7BN7","schema_version":"1.0","canonical_sha256":"b126cf85bf963e08aa2e2308a7b6ba499bafb39f53eaf9d9b1fd24bd3527ce67","source":{"kind":"arxiv","id":"2306.02405","version":1},"attestation_state":"computed","paper":{"title":"An Information-Theoretic Analysis of Self-supervised Discrete Representations of Speech","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Badr M. Abdullah, Bernd M\\\"obius, Dietrich Klakow, Mohammed Maqsood Shaik","submitted_at":"2023-06-04T16:52:11Z","abstract_excerpt":"Self-supervised representation learning for speech often involves a quantization step that transforms the acoustic input into discrete units. However, it remains unclear how to characterize the relationship between these discrete units and abstract phonetic categories such as phonemes. In this paper, we develop an information-theoretic framework whereby we represent each phonetic category as a distribution over discrete units. We then apply our framework to two different self-supervised models (namely wav2vec 2.0 and XLSR) and use American English speech as a case study. Our study demonstrates"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.02405","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-06-04T16:52:11Z","cross_cats_sorted":[],"title_canon_sha256":"1a987aaf41098123cb0be164fa9dd44721048d478ae59f7397de075df6cd975d","abstract_canon_sha256":"75d109728530734f94124ee6a940862fc3877804c3d02470e169e74909944e33"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:17:16.877713Z","signature_b64":"8Tcai2kjxINvcs7CY29CO0w0dPs2ekE0zAR/a3UpXYBAJ6veC9wbMtxKuGRqnK3kiaOwzzS5Uzmr3sQyT5bZDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b126cf85bf963e08aa2e2308a7b6ba499bafb39f53eaf9d9b1fd24bd3527ce67","last_reissued_at":"2026-07-05T06:17:16.877280Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:17:16.877280Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An Information-Theoretic Analysis of Self-supervised Discrete Representations of Speech","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Badr M. Abdullah, Bernd M\\\"obius, Dietrich Klakow, Mohammed Maqsood Shaik","submitted_at":"2023-06-04T16:52:11Z","abstract_excerpt":"Self-supervised representation learning for speech often involves a quantization step that transforms the acoustic input into discrete units. However, it remains unclear how to characterize the relationship between these discrete units and abstract phonetic categories such as phonemes. In this paper, we develop an information-theoretic framework whereby we represent each phonetic category as a distribution over discrete units. We then apply our framework to two different self-supervised models (namely wav2vec 2.0 and XLSR) and use American English speech as a case study. Our study demonstrates"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.02405","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.02405/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.02405","created_at":"2026-07-05T06:17:16.877341+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.02405v1","created_at":"2026-07-05T06:17:16.877341+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.02405","created_at":"2026-07-05T06:17:16.877341+00:00"},{"alias_kind":"pith_short_12","alias_value":"WETM7BN7SY7A","created_at":"2026-07-05T06:17:16.877341+00:00"},{"alias_kind":"pith_short_16","alias_value":"WETM7BN7SY7ARKRO","created_at":"2026-07-05T06:17:16.877341+00:00"},{"alias_kind":"pith_short_8","alias_value":"WETM7BN7","created_at":"2026-07-05T06:17:16.877341+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.16127","citing_title":"Improved Intelligibility of Dysarthric Speech using Conditional Flow Matching","ref_index":28,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WETM7BN7SY7ARKROEMEKPNV2JG","json":"https://pith.science/pith/WETM7BN7SY7ARKROEMEKPNV2JG.json","graph_json":"https://pith.science/api/pith-number/WETM7BN7SY7ARKROEMEKPNV2JG/graph.json","events_json":"https://pith.science/api/pith-number/WETM7BN7SY7ARKROEMEKPNV2JG/events.json","paper":"https://pith.science/paper/WETM7BN7"},"agent_actions":{"view_html":"https://pith.science/pith/WETM7BN7SY7ARKROEMEKPNV2JG","download_json":"https://pith.science/pith/WETM7BN7SY7ARKROEMEKPNV2JG.json","view_paper":"https://pith.science/paper/WETM7BN7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.02405&json=true","fetch_graph":"https://pith.science/api/pith-number/WETM7BN7SY7ARKROEMEKPNV2JG/graph.json","fetch_events":"https://pith.science/api/pith-number/WETM7BN7SY7ARKROEMEKPNV2JG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WETM7BN7SY7ARKROEMEKPNV2JG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WETM7BN7SY7ARKROEMEKPNV2JG/action/storage_attestation","attest_author":"https://pith.science/pith/WETM7BN7SY7ARKROEMEKPNV2JG/action/author_attestation","sign_citation":"https://pith.science/pith/WETM7BN7SY7ARKROEMEKPNV2JG/action/citation_signature","submit_replication":"https://pith.science/pith/WETM7BN7SY7ARKROEMEKPNV2JG/action/replication_record"}},"created_at":"2026-07-05T06:17:16.877341+00:00","updated_at":"2026-07-05T06:17:16.877341+00:00"}