{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:D4UXIJRT7X3NNNBVNU6X47P4BW","short_pith_number":"pith:D4UXIJRT","schema_version":"1.0","canonical_sha256":"1f29742633fdf6d6b4356d3d7e7dfc0dab77a7ec6f2fc1c5e5e8d2c609974d21","source":{"kind":"arxiv","id":"2304.10158","version":1},"attestation_state":"computed","paper":{"title":"Does Manipulating Tokenization Aid Cross-Lingual Transfer? A Study on POS Tagging for Non-Standardized Languages","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Barbara Plank, Hinrich Sch\\\"utze, Verena Blaschke","submitted_at":"2023-04-20T08:32:34Z","abstract_excerpt":"One of the challenges with finetuning pretrained language models (PLMs) is that their tokenizer is optimized for the language(s) it was pretrained on, but brittle when it comes to previously unseen variations in the data. This can for instance be observed when finetuning PLMs on one language and evaluating them on data in a closely related language variety with no standardized orthography. Despite the high linguistic similarity, tokenization no longer corresponds to meaningful representations of the target data, leading to low performance in, e.g., part-of-speech tagging.\n  In this work, we fi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.10158","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-04-20T08:32:34Z","cross_cats_sorted":[],"title_canon_sha256":"037dfa48788abce17957c037e773815696a5c0590d100dc101f8cdb3eb888de0","abstract_canon_sha256":"322afed52c728203786bbd871a613e947a7ec78d4b262b6a804a9262b5e31ec5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:02:54.663350Z","signature_b64":"iymmi+6VIpER9Ui2BMnzyIwx0/Z6y0ZgbVCcLFXfEAyBHFihE2uU4YVaJyMHdCBv42sMkqiCNhzEnMpHRJKDDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1f29742633fdf6d6b4356d3d7e7dfc0dab77a7ec6f2fc1c5e5e8d2c609974d21","last_reissued_at":"2026-07-05T06:02:54.662875Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:02:54.662875Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Does Manipulating Tokenization Aid Cross-Lingual Transfer? A Study on POS Tagging for Non-Standardized Languages","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Barbara Plank, Hinrich Sch\\\"utze, Verena Blaschke","submitted_at":"2023-04-20T08:32:34Z","abstract_excerpt":"One of the challenges with finetuning pretrained language models (PLMs) is that their tokenizer is optimized for the language(s) it was pretrained on, but brittle when it comes to previously unseen variations in the data. This can for instance be observed when finetuning PLMs on one language and evaluating them on data in a closely related language variety with no standardized orthography. Despite the high linguistic similarity, tokenization no longer corresponds to meaningful representations of the target data, leading to low performance in, e.g., part-of-speech tagging.\n  In this work, we fi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.10158","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.10158/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.10158","created_at":"2026-07-05T06:02:54.662943+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.10158v1","created_at":"2026-07-05T06:02:54.662943+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.10158","created_at":"2026-07-05T06:02:54.662943+00:00"},{"alias_kind":"pith_short_12","alias_value":"D4UXIJRT7X3N","created_at":"2026-07-05T06:02:54.662943+00:00"},{"alias_kind":"pith_short_16","alias_value":"D4UXIJRT7X3NNNBV","created_at":"2026-07-05T06:02:54.662943+00:00"},{"alias_kind":"pith_short_8","alias_value":"D4UXIJRT","created_at":"2026-07-05T06:02:54.662943+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.17715","citing_title":"Unveiling Factors for Enhanced POS Tagging: A Study of Low-Resource Medieval Romance Languages","ref_index":9,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/D4UXIJRT7X3NNNBVNU6X47P4BW","json":"https://pith.science/pith/D4UXIJRT7X3NNNBVNU6X47P4BW.json","graph_json":"https://pith.science/api/pith-number/D4UXIJRT7X3NNNBVNU6X47P4BW/graph.json","events_json":"https://pith.science/api/pith-number/D4UXIJRT7X3NNNBVNU6X47P4BW/events.json","paper":"https://pith.science/paper/D4UXIJRT"},"agent_actions":{"view_html":"https://pith.science/pith/D4UXIJRT7X3NNNBVNU6X47P4BW","download_json":"https://pith.science/pith/D4UXIJRT7X3NNNBVNU6X47P4BW.json","view_paper":"https://pith.science/paper/D4UXIJRT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.10158&json=true","fetch_graph":"https://pith.science/api/pith-number/D4UXIJRT7X3NNNBVNU6X47P4BW/graph.json","fetch_events":"https://pith.science/api/pith-number/D4UXIJRT7X3NNNBVNU6X47P4BW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/D4UXIJRT7X3NNNBVNU6X47P4BW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/D4UXIJRT7X3NNNBVNU6X47P4BW/action/storage_attestation","attest_author":"https://pith.science/pith/D4UXIJRT7X3NNNBVNU6X47P4BW/action/author_attestation","sign_citation":"https://pith.science/pith/D4UXIJRT7X3NNNBVNU6X47P4BW/action/citation_signature","submit_replication":"https://pith.science/pith/D4UXIJRT7X3NNNBVNU6X47P4BW/action/replication_record"}},"created_at":"2026-07-05T06:02:54.662943+00:00","updated_at":"2026-07-05T06:02:54.662943+00:00"}