{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:VRL6PETRGHU5GJEBBHPPCFUZ7J","short_pith_number":"pith:VRL6PETR","schema_version":"1.0","canonical_sha256":"ac57e7927131e9d3248109def11699fa4fd754d3614ab800e8ce53fe47cef126","source":{"kind":"arxiv","id":"2407.18332","version":1},"attestation_state":"computed","paper":{"title":"Analyzing Speech Unit Selection for Textless Speech-to-Speech Translation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG","cs.SD"],"primary_cat":"eess.AS","authors_text":"Jarod Duret (LIA), Titouan Parcollet (CAM), Yannick Est\\`eve (LIA)","submitted_at":"2024-07-08T08:53:26Z","abstract_excerpt":"Recent advancements in textless speech-to-speech translation systems have been driven by the adoption of self-supervised learning techniques.     Although most state-of-the-art systems adopt a similar architecture to transform source language speech into sequences of discrete representations in the target language, the criteria for selecting these target speech units remains an open question.    This work explores the selection process through a study of downstream tasks such as automatic speech recognition, speech synthesis, speaker recognition, and emotion recognition.      Interestingly, ou"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.18332","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2024-07-08T08:53:26Z","cross_cats_sorted":["cs.CL","cs.LG","cs.SD"],"title_canon_sha256":"cef13adc7cc836e20bf11a4c1562448fe5de707b651ace580942033027b93eee","abstract_canon_sha256":"00dc547c1ddc63a463ae5033a3b17f492d53fa32ad5abaf5090bc0a39a6dea1a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:48:47.970718Z","signature_b64":"IM5k4F6prH5sXbOPjucHg7v0pOBgEjHLyPrzEkrTJ/P56vgwRbDwcueMKl1lk5WdgZXAofOVyaEVmaCuh4FEBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ac57e7927131e9d3248109def11699fa4fd754d3614ab800e8ce53fe47cef126","last_reissued_at":"2026-07-05T08:48:47.970229Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:48:47.970229Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Analyzing Speech Unit Selection for Textless Speech-to-Speech Translation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG","cs.SD"],"primary_cat":"eess.AS","authors_text":"Jarod Duret (LIA), Titouan Parcollet (CAM), Yannick Est\\`eve (LIA)","submitted_at":"2024-07-08T08:53:26Z","abstract_excerpt":"Recent advancements in textless speech-to-speech translation systems have been driven by the adoption of self-supervised learning techniques.     Although most state-of-the-art systems adopt a similar architecture to transform source language speech into sequences of discrete representations in the target language, the criteria for selecting these target speech units remains an open question.    This work explores the selection process through a study of downstream tasks such as automatic speech recognition, speech synthesis, speaker recognition, and emotion recognition.      Interestingly, ou"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.18332","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.18332/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.18332","created_at":"2026-07-05T08:48:47.970284+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.18332v1","created_at":"2026-07-05T08:48:47.970284+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.18332","created_at":"2026-07-05T08:48:47.970284+00:00"},{"alias_kind":"pith_short_12","alias_value":"VRL6PETRGHU5","created_at":"2026-07-05T08:48:47.970284+00:00"},{"alias_kind":"pith_short_16","alias_value":"VRL6PETRGHU5GJEB","created_at":"2026-07-05T08:48:47.970284+00:00"},{"alias_kind":"pith_short_8","alias_value":"VRL6PETR","created_at":"2026-07-05T08:48:47.970284+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.02443","citing_title":"Breaking the Barriers of Text-Hungry and Audio-Deficient AI","ref_index":30,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VRL6PETRGHU5GJEBBHPPCFUZ7J","json":"https://pith.science/pith/VRL6PETRGHU5GJEBBHPPCFUZ7J.json","graph_json":"https://pith.science/api/pith-number/VRL6PETRGHU5GJEBBHPPCFUZ7J/graph.json","events_json":"https://pith.science/api/pith-number/VRL6PETRGHU5GJEBBHPPCFUZ7J/events.json","paper":"https://pith.science/paper/VRL6PETR"},"agent_actions":{"view_html":"https://pith.science/pith/VRL6PETRGHU5GJEBBHPPCFUZ7J","download_json":"https://pith.science/pith/VRL6PETRGHU5GJEBBHPPCFUZ7J.json","view_paper":"https://pith.science/paper/VRL6PETR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.18332&json=true","fetch_graph":"https://pith.science/api/pith-number/VRL6PETRGHU5GJEBBHPPCFUZ7J/graph.json","fetch_events":"https://pith.science/api/pith-number/VRL6PETRGHU5GJEBBHPPCFUZ7J/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VRL6PETRGHU5GJEBBHPPCFUZ7J/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VRL6PETRGHU5GJEBBHPPCFUZ7J/action/storage_attestation","attest_author":"https://pith.science/pith/VRL6PETRGHU5GJEBBHPPCFUZ7J/action/author_attestation","sign_citation":"https://pith.science/pith/VRL6PETRGHU5GJEBBHPPCFUZ7J/action/citation_signature","submit_replication":"https://pith.science/pith/VRL6PETRGHU5GJEBBHPPCFUZ7J/action/replication_record"}},"created_at":"2026-07-05T08:48:47.970284+00:00","updated_at":"2026-07-05T08:48:47.970284+00:00"}