{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:OFIOFJXDFDYUKRN67TVLJPWLXN","short_pith_number":"pith:OFIOFJXD","schema_version":"1.0","canonical_sha256":"7150e2a6e328f14545befceab4becbbb4adaf7edf14d4d1ee3a68cc02051c712","source":{"kind":"arxiv","id":"2504.16977","version":1},"attestation_state":"computed","paper":{"title":"Tokenization Matters: Improving Zero-Shot NER for Indic Languages","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Amit Agarwal, Hitesh Laxmichand Patel, Priyaranjan Pattnayak","submitted_at":"2025-04-23T17:28:38Z","abstract_excerpt":"Tokenization is a critical component of Natural Language Processing (NLP), especially for low resource languages, where subword segmentation influences vocabulary structure and downstream task accuracy. Although Byte Pair Encoding (BPE) is a standard tokenization method in multilingual language models, its suitability for Named Entity Recognition (NER) in low resource Indic languages remains underexplored due to its limitations in handling morphological complexity. In this work, we systematically compare BPE, SentencePiece, and Character Level tokenization strategies using IndicBERT for NER ta"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.16977","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-04-23T17:28:38Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"9b750e8fc04fad5d6f0fd69d36f9afa1465949a6e0c4944f5850d02886eca59f","abstract_canon_sha256":"2b9a43277938287e2ed06c7544a53e1c105e2682d009c550ce5714a4d5b1450e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:53:20.070874Z","signature_b64":"ANmpXAeyiXg3Bmn/bIJkRshoqlDpxVqojAmg4nOZlhcb4WfRF2KfoohSvVfPmdDmJxPmpSpeZ3TkOzx96CPPCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7150e2a6e328f14545befceab4becbbb4adaf7edf14d4d1ee3a68cc02051c712","last_reissued_at":"2026-07-05T10:53:20.070411Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:53:20.070411Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Tokenization Matters: Improving Zero-Shot NER for Indic Languages","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Amit Agarwal, Hitesh Laxmichand Patel, Priyaranjan Pattnayak","submitted_at":"2025-04-23T17:28:38Z","abstract_excerpt":"Tokenization is a critical component of Natural Language Processing (NLP), especially for low resource languages, where subword segmentation influences vocabulary structure and downstream task accuracy. Although Byte Pair Encoding (BPE) is a standard tokenization method in multilingual language models, its suitability for Named Entity Recognition (NER) in low resource Indic languages remains underexplored due to its limitations in handling morphological complexity. In this work, we systematically compare BPE, SentencePiece, and Character Level tokenization strategies using IndicBERT for NER ta"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.16977","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.16977/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.16977","created_at":"2026-07-05T10:53:20.070472+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.16977v1","created_at":"2026-07-05T10:53:20.070472+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.16977","created_at":"2026-07-05T10:53:20.070472+00:00"},{"alias_kind":"pith_short_12","alias_value":"OFIOFJXDFDYU","created_at":"2026-07-05T10:53:20.070472+00:00"},{"alias_kind":"pith_short_16","alias_value":"OFIOFJXDFDYUKRN6","created_at":"2026-07-05T10:53:20.070472+00:00"},{"alias_kind":"pith_short_8","alias_value":"OFIOFJXD","created_at":"2026-07-05T10:53:20.070472+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.02097","citing_title":"Hybrid AI for Responsive Multi-Turn Online Conversations with Novel Dynamic Routing and Feedback Adaptation","ref_index":28,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OFIOFJXDFDYUKRN67TVLJPWLXN","json":"https://pith.science/pith/OFIOFJXDFDYUKRN67TVLJPWLXN.json","graph_json":"https://pith.science/api/pith-number/OFIOFJXDFDYUKRN67TVLJPWLXN/graph.json","events_json":"https://pith.science/api/pith-number/OFIOFJXDFDYUKRN67TVLJPWLXN/events.json","paper":"https://pith.science/paper/OFIOFJXD"},"agent_actions":{"view_html":"https://pith.science/pith/OFIOFJXDFDYUKRN67TVLJPWLXN","download_json":"https://pith.science/pith/OFIOFJXDFDYUKRN67TVLJPWLXN.json","view_paper":"https://pith.science/paper/OFIOFJXD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.16977&json=true","fetch_graph":"https://pith.science/api/pith-number/OFIOFJXDFDYUKRN67TVLJPWLXN/graph.json","fetch_events":"https://pith.science/api/pith-number/OFIOFJXDFDYUKRN67TVLJPWLXN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OFIOFJXDFDYUKRN67TVLJPWLXN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OFIOFJXDFDYUKRN67TVLJPWLXN/action/storage_attestation","attest_author":"https://pith.science/pith/OFIOFJXDFDYUKRN67TVLJPWLXN/action/author_attestation","sign_citation":"https://pith.science/pith/OFIOFJXDFDYUKRN67TVLJPWLXN/action/citation_signature","submit_replication":"https://pith.science/pith/OFIOFJXDFDYUKRN67TVLJPWLXN/action/replication_record"}},"created_at":"2026-07-05T10:53:20.070472+00:00","updated_at":"2026-07-05T10:53:20.070472+00:00"}