{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:THBLHXNX3PCCTX7P6LJ22OCIMX","short_pith_number":"pith:THBLHXNX","schema_version":"1.0","canonical_sha256":"99c2b3ddb7dbc429dfeff2d3ad384865ddb2c0a5e22d20ea2dc0e651f14db4ed","source":{"kind":"arxiv","id":"1912.02481","version":2},"attestation_state":"computed","paper":{"title":"Massive vs. Curated Word Embeddings for Low-Resourced Languages. The Case of Yor\\`ub\\'a and Twi","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Cristina Espa\\~na-Bonet, David I. Adelani, Jesujoba O. Alabi, Kwabena Amponsah-Kaakyire","submitted_at":"2019-12-05T10:25:32Z","abstract_excerpt":"The success of several architectures to learn semantic representations from unannotated text and the availability of these kind of texts in online multilingual resources such as Wikipedia has facilitated the massive and automatic creation of resources for multiple languages. The evaluation of such resources is usually done for the high-resourced languages, where one has a smorgasbord of tasks and test sets to evaluate on. For low-resourced languages, the evaluation is more difficult and normally ignored, with the hope that the impressive capability of deep learning architectures to learn (mult"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1912.02481","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2019-12-05T10:25:32Z","cross_cats_sorted":[],"title_canon_sha256":"642f48d041058b37e52dd66850cc18427b058a4667a4164efa3c5b3aa798dd0e","abstract_canon_sha256":"f296142b7de98f013c9b5167ba1ad7d65c8960cfea826f193ded8656505fe108"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:51:05.121155Z","signature_b64":"X7jpKUNVZau3CrrUusPJPjm5YRv4wMOPExP6llxRVQMdNOqH5JO9BqVajdilY15rHPT7S979dpUfGzXh4m93CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"99c2b3ddb7dbc429dfeff2d3ad384865ddb2c0a5e22d20ea2dc0e651f14db4ed","last_reissued_at":"2026-07-05T00:51:05.120618Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:51:05.120618Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Massive vs. Curated Word Embeddings for Low-Resourced Languages. The Case of Yor\\`ub\\'a and Twi","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Cristina Espa\\~na-Bonet, David I. Adelani, Jesujoba O. Alabi, Kwabena Amponsah-Kaakyire","submitted_at":"2019-12-05T10:25:32Z","abstract_excerpt":"The success of several architectures to learn semantic representations from unannotated text and the availability of these kind of texts in online multilingual resources such as Wikipedia has facilitated the massive and automatic creation of resources for multiple languages. The evaluation of such resources is usually done for the high-resourced languages, where one has a smorgasbord of tasks and test sets to evaluate on. For low-resourced languages, the evaluation is more difficult and normally ignored, with the hope that the impressive capability of deep learning architectures to learn (mult"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1912.02481","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1912.02481/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1912.02481","created_at":"2026-07-05T00:51:05.120680+00:00"},{"alias_kind":"arxiv_version","alias_value":"1912.02481v2","created_at":"2026-07-05T00:51:05.120680+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1912.02481","created_at":"2026-07-05T00:51:05.120680+00:00"},{"alias_kind":"pith_short_12","alias_value":"THBLHXNX3PCC","created_at":"2026-07-05T00:51:05.120680+00:00"},{"alias_kind":"pith_short_16","alias_value":"THBLHXNX3PCCTX7P","created_at":"2026-07-05T00:51:05.120680+00:00"},{"alias_kind":"pith_short_8","alias_value":"THBLHXNX","created_at":"2026-07-05T00:51:05.120680+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.13020","citing_title":"Edeflip: Supervised Word Translation between English and Yoruba","ref_index":1,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/THBLHXNX3PCCTX7P6LJ22OCIMX","json":"https://pith.science/pith/THBLHXNX3PCCTX7P6LJ22OCIMX.json","graph_json":"https://pith.science/api/pith-number/THBLHXNX3PCCTX7P6LJ22OCIMX/graph.json","events_json":"https://pith.science/api/pith-number/THBLHXNX3PCCTX7P6LJ22OCIMX/events.json","paper":"https://pith.science/paper/THBLHXNX"},"agent_actions":{"view_html":"https://pith.science/pith/THBLHXNX3PCCTX7P6LJ22OCIMX","download_json":"https://pith.science/pith/THBLHXNX3PCCTX7P6LJ22OCIMX.json","view_paper":"https://pith.science/paper/THBLHXNX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1912.02481&json=true","fetch_graph":"https://pith.science/api/pith-number/THBLHXNX3PCCTX7P6LJ22OCIMX/graph.json","fetch_events":"https://pith.science/api/pith-number/THBLHXNX3PCCTX7P6LJ22OCIMX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/THBLHXNX3PCCTX7P6LJ22OCIMX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/THBLHXNX3PCCTX7P6LJ22OCIMX/action/storage_attestation","attest_author":"https://pith.science/pith/THBLHXNX3PCCTX7P6LJ22OCIMX/action/author_attestation","sign_citation":"https://pith.science/pith/THBLHXNX3PCCTX7P6LJ22OCIMX/action/citation_signature","submit_replication":"https://pith.science/pith/THBLHXNX3PCCTX7P6LJ22OCIMX/action/replication_record"}},"created_at":"2026-07-05T00:51:05.120680+00:00","updated_at":"2026-07-05T00:51:05.120680+00:00"}