{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:V4CLKNR7G5D34I33GRS3BPJBW2","short_pith_number":"pith:V4CLKNR7","schema_version":"1.0","canonical_sha256":"af04b5363f3747be237b3465b0bd21b6a764c33213b91d2ae102906cecdb1b84","source":{"kind":"arxiv","id":"2406.19564","version":1},"attestation_state":"computed","paper":{"title":"Voices Unheard: NLP Resources and Models for Yor\\`ub\\'a Regional Dialects","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Anuoluwapo Aremu, Daud Abolade, David Ifeoluwa Adelani, Diana Abagyan, Hila Gonen, Noah A. Smith, Orevaoghene Ahia, Yulia Tsvetkov","submitted_at":"2024-06-27T22:38:04Z","abstract_excerpt":"Yor\\`ub\\'a an African language with roughly 47 million speakers encompasses a continuum with several dialects. Recent efforts to develop NLP technologies for African languages have focused on their standard dialects, resulting in disparities for dialects and varieties for which there are little to no resources or tools. We take steps towards bridging this gap by introducing a new high-quality parallel text and speech corpus YOR\\`ULECT across three domains and four regional Yor\\`ub\\'a dialects. To develop this corpus, we engaged native speakers, travelling to communities where these dialects ar"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.19564","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-27T22:38:04Z","cross_cats_sorted":[],"title_canon_sha256":"c72ef1e3eed96391c813eb5c393a1945e3ecef1a4616f5ff4ce105b564aacf3d","abstract_canon_sha256":"1f9bb1f909574f98e72b09a7232954c8736157f21395bf687b4ebd3d1ad3ee4a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:37:41.872114Z","signature_b64":"LO8snp7qABbVt5kVHyhXCUxHZh6I3VF6QapP58YlBWWEo5T9vPl3va3o44TGLO3VR3gORCK0Etbbrsuq12bbDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"af04b5363f3747be237b3465b0bd21b6a764c33213b91d2ae102906cecdb1b84","last_reissued_at":"2026-07-05T08:37:41.871670Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:37:41.871670Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Voices Unheard: NLP Resources and Models for Yor\\`ub\\'a Regional Dialects","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Anuoluwapo Aremu, Daud Abolade, David Ifeoluwa Adelani, Diana Abagyan, Hila Gonen, Noah A. Smith, Orevaoghene Ahia, Yulia Tsvetkov","submitted_at":"2024-06-27T22:38:04Z","abstract_excerpt":"Yor\\`ub\\'a an African language with roughly 47 million speakers encompasses a continuum with several dialects. Recent efforts to develop NLP technologies for African languages have focused on their standard dialects, resulting in disparities for dialects and varieties for which there are little to no resources or tools. We take steps towards bridging this gap by introducing a new high-quality parallel text and speech corpus YOR\\`ULECT across three domains and four regional Yor\\`ub\\'a dialects. To develop this corpus, we engaged native speakers, travelling to communities where these dialects ar"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.19564","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.19564/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.19564","created_at":"2026-07-05T08:37:41.871729+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.19564v1","created_at":"2026-07-05T08:37:41.871729+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.19564","created_at":"2026-07-05T08:37:41.871729+00:00"},{"alias_kind":"pith_short_12","alias_value":"V4CLKNR7G5D3","created_at":"2026-07-05T08:37:41.871729+00:00"},{"alias_kind":"pith_short_16","alias_value":"V4CLKNR7G5D34I33","created_at":"2026-07-05T08:37:41.871729+00:00"},{"alias_kind":"pith_short_8","alias_value":"V4CLKNR7","created_at":"2026-07-05T08:37:41.871729+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.21563","citing_title":"FormosanBench: Benchmarking Low-Resource Austronesian Languages in the Era of Large Language Models","ref_index":3,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/V4CLKNR7G5D34I33GRS3BPJBW2","json":"https://pith.science/pith/V4CLKNR7G5D34I33GRS3BPJBW2.json","graph_json":"https://pith.science/api/pith-number/V4CLKNR7G5D34I33GRS3BPJBW2/graph.json","events_json":"https://pith.science/api/pith-number/V4CLKNR7G5D34I33GRS3BPJBW2/events.json","paper":"https://pith.science/paper/V4CLKNR7"},"agent_actions":{"view_html":"https://pith.science/pith/V4CLKNR7G5D34I33GRS3BPJBW2","download_json":"https://pith.science/pith/V4CLKNR7G5D34I33GRS3BPJBW2.json","view_paper":"https://pith.science/paper/V4CLKNR7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.19564&json=true","fetch_graph":"https://pith.science/api/pith-number/V4CLKNR7G5D34I33GRS3BPJBW2/graph.json","fetch_events":"https://pith.science/api/pith-number/V4CLKNR7G5D34I33GRS3BPJBW2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/V4CLKNR7G5D34I33GRS3BPJBW2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/V4CLKNR7G5D34I33GRS3BPJBW2/action/storage_attestation","attest_author":"https://pith.science/pith/V4CLKNR7G5D34I33GRS3BPJBW2/action/author_attestation","sign_citation":"https://pith.science/pith/V4CLKNR7G5D34I33GRS3BPJBW2/action/citation_signature","submit_replication":"https://pith.science/pith/V4CLKNR7G5D34I33GRS3BPJBW2/action/replication_record"}},"created_at":"2026-07-05T08:37:41.871729+00:00","updated_at":"2026-07-05T08:37:41.871729+00:00"}