{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:BZORYYHYHU6IAG2EKV2MJYIXT7","short_pith_number":"pith:BZORYYHY","schema_version":"1.0","canonical_sha256":"0e5d1c60f83d3c801b445574c4e1179ff3a6b52804f355ff418abc3e8823f7dc","source":{"kind":"arxiv","id":"2607.23319","version":1},"attestation_state":"computed","paper":{"title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CY","cs.ET","cs.LG"],"primary_cat":"cs.CL","authors_text":"Lakshmi Rajendran, Pavithra Muruganantham, Poornima Kumaresan, Santhosh Sivasubramani","submitted_at":"2026-07-25T18:23:06Z","abstract_excerpt":"Standard subword tokenization algorithms such as Byte-Pair Encoding (BPE) and SentencePiece are trained predominantly on modern language corpora and produce inefficient segmentations when applied to classical Indian languages. Sanskrit, Tamil, and other classical Indic languages exhibit agglutinative morphology, productive sandhi (phonological fusion at word boundaries), and domain-specific vocabularies absent from general-purpose training data. This paper presents BHARATI, a set of SentencePiece BPE tokenizers trained on a balanced 781 MB corpus spanning seven languages (English, Hindi, Sansk"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.23319","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-07-25T18:23:06Z","cross_cats_sorted":["cs.CY","cs.ET","cs.LG"],"title_canon_sha256":"51c401fc6775437ac04c675b473d80413abdea2fb1207e7dac1f43a27cc7823c","abstract_canon_sha256":"bb6f2c19b89a7104791febdcb18d512c99b44a67f15bfd26bdce09c651b1476d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-28T01:22:47.537964Z","signature_b64":"lqrr2JzCkXOIi3aMWOmAnpY59wXmGCiZT5M/bbFiCq2+XDTDZC9f3lDkjOeZyXsUbQVTMpC23QcqdldnLFIjBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0e5d1c60f83d3c801b445574c4e1179ff3a6b52804f355ff418abc3e8823f7dc","last_reissued_at":"2026-07-28T01:22:47.537198Z","signature_status":"signed_v1","first_computed_at":"2026-07-28T01:22:47.537198Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CY","cs.ET","cs.LG"],"primary_cat":"cs.CL","authors_text":"Lakshmi Rajendran, Pavithra Muruganantham, Poornima Kumaresan, Santhosh Sivasubramani","submitted_at":"2026-07-25T18:23:06Z","abstract_excerpt":"Standard subword tokenization algorithms such as Byte-Pair Encoding (BPE) and SentencePiece are trained predominantly on modern language corpora and produce inefficient segmentations when applied to classical Indian languages. Sanskrit, Tamil, and other classical Indic languages exhibit agglutinative morphology, productive sandhi (phonological fusion at word boundaries), and domain-specific vocabularies absent from general-purpose training data. This paper presents BHARATI, a set of SentencePiece BPE tokenizers trained on a balanced 781 MB corpus spanning seven languages (English, Hindi, Sansk"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.23319","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.23319/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.23319","created_at":"2026-07-28T01:22:47.537607+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.23319v1","created_at":"2026-07-28T01:22:47.537607+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.23319","created_at":"2026-07-28T01:22:47.537607+00:00"},{"alias_kind":"pith_short_12","alias_value":"BZORYYHYHU6I","created_at":"2026-07-28T01:22:47.537607+00:00"},{"alias_kind":"pith_short_16","alias_value":"BZORYYHYHU6IAG2E","created_at":"2026-07-28T01:22:47.537607+00:00"},{"alias_kind":"pith_short_8","alias_value":"BZORYYHY","created_at":"2026-07-28T01:22:47.537607+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BZORYYHYHU6IAG2EKV2MJYIXT7","json":"https://pith.science/pith/BZORYYHYHU6IAG2EKV2MJYIXT7.json","graph_json":"https://pith.science/api/pith-number/BZORYYHYHU6IAG2EKV2MJYIXT7/graph.json","events_json":"https://pith.science/api/pith-number/BZORYYHYHU6IAG2EKV2MJYIXT7/events.json","paper":"https://pith.science/paper/BZORYYHY"},"agent_actions":{"view_html":"https://pith.science/pith/BZORYYHYHU6IAG2EKV2MJYIXT7","download_json":"https://pith.science/pith/BZORYYHYHU6IAG2EKV2MJYIXT7.json","view_paper":"https://pith.science/paper/BZORYYHY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.23319&json=true","fetch_graph":"https://pith.science/api/pith-number/BZORYYHYHU6IAG2EKV2MJYIXT7/graph.json","fetch_events":"https://pith.science/api/pith-number/BZORYYHYHU6IAG2EKV2MJYIXT7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BZORYYHYHU6IAG2EKV2MJYIXT7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BZORYYHYHU6IAG2EKV2MJYIXT7/action/storage_attestation","attest_author":"https://pith.science/pith/BZORYYHYHU6IAG2EKV2MJYIXT7/action/author_attestation","sign_citation":"https://pith.science/pith/BZORYYHYHU6IAG2EKV2MJYIXT7/action/citation_signature","submit_replication":"https://pith.science/pith/BZORYYHYHU6IAG2EKV2MJYIXT7/action/replication_record"}},"created_at":"2026-07-28T01:22:47.537607+00:00","updated_at":"2026-07-28T01:22:47.537607+00:00"}