{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:K2FJF4NUK4WBW5WIGXUEX4M4ZK","short_pith_number":"pith:K2FJF4NU","schema_version":"1.0","canonical_sha256":"568a92f1b4572c1b76c835e84bf19cca93b94f258680b58b069f8cac6f453652","source":{"kind":"arxiv","id":"2203.16825","version":1},"attestation_state":"computed","paper":{"title":"indic-punct: An automatic punctuation restoration and inverse text normalization framework for Indic languages","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Anirudh Gupta, Ankur Dhuriya, Harveen Singh Chadha, Neeraj Chhimwal, Priyanshi Shah, Rishabh Gaur, Vivek Raghavan","submitted_at":"2022-03-31T06:18:43Z","abstract_excerpt":"Automatic Speech Recognition (ASR) generates text which is most of the times devoid of any punctuation. Absence of punctuation is text can affect readability. Also, down stream NLP tasks such as sentiment analysis, machine translation, greatly benefit by having punctuation and sentence boundary information. We present an approach for automatic punctuation of text using a pretrained IndicBERT model. Inverse text normalization is done by hand writing weighted finite state transducer (WFST) grammars. We have developed this tool for 11 Indic languages namely Hindi, Tamil, Telugu, Kannada, Gujarati"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2203.16825","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-03-31T06:18:43Z","cross_cats_sorted":[],"title_canon_sha256":"4a9f7351a001c8bbf974a583a2edb4a769d0686ae98cd4d224658206cc6882b7","abstract_canon_sha256":"6d9cb9d9bcdb814b3ffa24131a6afa88b39c7439152efa4280fba9647e0601bd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:10:28.431533Z","signature_b64":"/2YJjbR82wXbCsl+kXLRE7W5Tp3yWuJsrznnNBnpBU57fI1nD3+sHfIy2fHacp4Le4LssHE3YSEZpuELpb88CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"568a92f1b4572c1b76c835e84bf19cca93b94f258680b58b069f8cac6f453652","last_reissued_at":"2026-07-05T04:10:28.431059Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:10:28.431059Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"indic-punct: An automatic punctuation restoration and inverse text normalization framework for Indic languages","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Anirudh Gupta, Ankur Dhuriya, Harveen Singh Chadha, Neeraj Chhimwal, Priyanshi Shah, Rishabh Gaur, Vivek Raghavan","submitted_at":"2022-03-31T06:18:43Z","abstract_excerpt":"Automatic Speech Recognition (ASR) generates text which is most of the times devoid of any punctuation. Absence of punctuation is text can affect readability. Also, down stream NLP tasks such as sentiment analysis, machine translation, greatly benefit by having punctuation and sentence boundary information. We present an approach for automatic punctuation of text using a pretrained IndicBERT model. Inverse text normalization is done by hand writing weighted finite state transducer (WFST) grammars. We have developed this tool for 11 Indic languages namely Hindi, Tamil, Telugu, Kannada, Gujarati"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2203.16825","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2203.16825/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2203.16825","created_at":"2026-07-05T04:10:28.431118+00:00"},{"alias_kind":"arxiv_version","alias_value":"2203.16825v1","created_at":"2026-07-05T04:10:28.431118+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2203.16825","created_at":"2026-07-05T04:10:28.431118+00:00"},{"alias_kind":"pith_short_12","alias_value":"K2FJF4NUK4WB","created_at":"2026-07-05T04:10:28.431118+00:00"},{"alias_kind":"pith_short_16","alias_value":"K2FJF4NUK4WBW5WI","created_at":"2026-07-05T04:10:28.431118+00:00"},{"alias_kind":"pith_short_8","alias_value":"K2FJF4NU","created_at":"2026-07-05T04:10:28.431118+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.18423","citing_title":"BhashaSutra: A Task-Centric Unified Survey of Indian NLP Datasets, Corpora, and Resources","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K2FJF4NUK4WBW5WIGXUEX4M4ZK","json":"https://pith.science/pith/K2FJF4NUK4WBW5WIGXUEX4M4ZK.json","graph_json":"https://pith.science/api/pith-number/K2FJF4NUK4WBW5WIGXUEX4M4ZK/graph.json","events_json":"https://pith.science/api/pith-number/K2FJF4NUK4WBW5WIGXUEX4M4ZK/events.json","paper":"https://pith.science/paper/K2FJF4NU"},"agent_actions":{"view_html":"https://pith.science/pith/K2FJF4NUK4WBW5WIGXUEX4M4ZK","download_json":"https://pith.science/pith/K2FJF4NUK4WBW5WIGXUEX4M4ZK.json","view_paper":"https://pith.science/paper/K2FJF4NU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2203.16825&json=true","fetch_graph":"https://pith.science/api/pith-number/K2FJF4NUK4WBW5WIGXUEX4M4ZK/graph.json","fetch_events":"https://pith.science/api/pith-number/K2FJF4NUK4WBW5WIGXUEX4M4ZK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K2FJF4NUK4WBW5WIGXUEX4M4ZK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K2FJF4NUK4WBW5WIGXUEX4M4ZK/action/storage_attestation","attest_author":"https://pith.science/pith/K2FJF4NUK4WBW5WIGXUEX4M4ZK/action/author_attestation","sign_citation":"https://pith.science/pith/K2FJF4NUK4WBW5WIGXUEX4M4ZK/action/citation_signature","submit_replication":"https://pith.science/pith/K2FJF4NUK4WBW5WIGXUEX4M4ZK/action/replication_record"}},"created_at":"2026-07-05T04:10:28.431118+00:00","updated_at":"2026-07-05T04:10:28.431118+00:00"}