{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:IT4NMINQNX2WR3IUKQS4X7VF3Y","short_pith_number":"pith:IT4NMINQ","schema_version":"1.0","canonical_sha256":"44f8d621b06df568ed145425cbfea5de24a75ceab5817288178008dd9f6cea2f","source":{"kind":"arxiv","id":"2109.04715","version":1},"attestation_state":"computed","paper":{"title":"AfroMT: Pretraining Strategies and Reproducible Benchmarks for Translation of 8 African Languages","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Graham Neubig, Junjie Hu, Machel Reid, Yutaka Matsuo","submitted_at":"2021-09-10T07:45:21Z","abstract_excerpt":"Reproducible benchmarks are crucial in driving progress of machine translation research. However, existing machine translation benchmarks have been mostly limited to high-resource or well-represented languages. Despite an increasing interest in low-resource machine translation, there are no standardized reproducible benchmarks for many African languages, many of which are used by millions of speakers but have less digitized textual data. To tackle these challenges, we propose AfroMT, a standardized, clean, and reproducible machine translation benchmark for eight widely spoken African languages"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2109.04715","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2021-09-10T07:45:21Z","cross_cats_sorted":[],"title_canon_sha256":"d95d163d90c3326b357fb8ec9455e6027b7d44c23aa6191c387e28e9e2204bd4","abstract_canon_sha256":"0b98fd58b7f65d2a536af36042de47731065caabfe14d635b08cf002eabbb375"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:13:09.292279Z","signature_b64":"zXIvDAVbYG3zV7UtsASJ3/GzXgrmDtWoDMZ5ocUfibDlLNfSiChytBCoW7LevLDtlJY4bt2sPXwQfiwTt5BFAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"44f8d621b06df568ed145425cbfea5de24a75ceab5817288178008dd9f6cea2f","last_reissued_at":"2026-07-05T03:13:09.291931Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:13:09.291931Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AfroMT: Pretraining Strategies and Reproducible Benchmarks for Translation of 8 African Languages","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Graham Neubig, Junjie Hu, Machel Reid, Yutaka Matsuo","submitted_at":"2021-09-10T07:45:21Z","abstract_excerpt":"Reproducible benchmarks are crucial in driving progress of machine translation research. However, existing machine translation benchmarks have been mostly limited to high-resource or well-represented languages. Despite an increasing interest in low-resource machine translation, there are no standardized reproducible benchmarks for many African languages, many of which are used by millions of speakers but have less digitized textual data. To tackle these challenges, we propose AfroMT, a standardized, clean, and reproducible machine translation benchmark for eight widely spoken African languages"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2109.04715","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2109.04715/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2109.04715","created_at":"2026-07-05T03:13:09.291993+00:00"},{"alias_kind":"arxiv_version","alias_value":"2109.04715v1","created_at":"2026-07-05T03:13:09.291993+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2109.04715","created_at":"2026-07-05T03:13:09.291993+00:00"},{"alias_kind":"pith_short_12","alias_value":"IT4NMINQNX2W","created_at":"2026-07-05T03:13:09.291993+00:00"},{"alias_kind":"pith_short_16","alias_value":"IT4NMINQNX2WR3IU","created_at":"2026-07-05T03:13:09.291993+00:00"},{"alias_kind":"pith_short_8","alias_value":"IT4NMINQ","created_at":"2026-07-05T03:13:09.291993+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.18436","citing_title":"Voice of a Continent: Mapping Africa's Speech Technology Frontier","ref_index":44,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IT4NMINQNX2WR3IUKQS4X7VF3Y","json":"https://pith.science/pith/IT4NMINQNX2WR3IUKQS4X7VF3Y.json","graph_json":"https://pith.science/api/pith-number/IT4NMINQNX2WR3IUKQS4X7VF3Y/graph.json","events_json":"https://pith.science/api/pith-number/IT4NMINQNX2WR3IUKQS4X7VF3Y/events.json","paper":"https://pith.science/paper/IT4NMINQ"},"agent_actions":{"view_html":"https://pith.science/pith/IT4NMINQNX2WR3IUKQS4X7VF3Y","download_json":"https://pith.science/pith/IT4NMINQNX2WR3IUKQS4X7VF3Y.json","view_paper":"https://pith.science/paper/IT4NMINQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2109.04715&json=true","fetch_graph":"https://pith.science/api/pith-number/IT4NMINQNX2WR3IUKQS4X7VF3Y/graph.json","fetch_events":"https://pith.science/api/pith-number/IT4NMINQNX2WR3IUKQS4X7VF3Y/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IT4NMINQNX2WR3IUKQS4X7VF3Y/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IT4NMINQNX2WR3IUKQS4X7VF3Y/action/storage_attestation","attest_author":"https://pith.science/pith/IT4NMINQNX2WR3IUKQS4X7VF3Y/action/author_attestation","sign_citation":"https://pith.science/pith/IT4NMINQNX2WR3IUKQS4X7VF3Y/action/citation_signature","submit_replication":"https://pith.science/pith/IT4NMINQNX2WR3IUKQS4X7VF3Y/action/replication_record"}},"created_at":"2026-07-05T03:13:09.291993+00:00","updated_at":"2026-07-05T03:13:09.291993+00:00"}