{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:YLGFVISZ5FDJ6W47ZRW2GJ2OH6","short_pith_number":"pith:YLGFVISZ","schema_version":"1.0","canonical_sha256":"c2cc5aa259e9469f5b9fcc6da3274e3fb3df0dcf9fb3f9844f04afb7aa899681","source":{"kind":"arxiv","id":"2112.13800","version":1},"attestation_state":"computed","paper":{"title":"\"A Passage to India\": Pre-trained Word Embeddings for Indian Languages","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Diptesh Kanojia, Kumar Saunack, Kumar Saurav, Pushpak Bhattacharyya","submitted_at":"2021-12-27T17:31:04Z","abstract_excerpt":"Dense word vectors or 'word embeddings' which encode semantic properties of words, have now become integral to NLP tasks like Machine Translation (MT), Question Answering (QA), Word Sense Disambiguation (WSD), and Information Retrieval (IR). In this paper, we use various existing approaches to create multiple word embeddings for 14 Indian languages. We place these embeddings for all these languages, viz., Assamese, Bengali, Gujarati, Hindi, Kannada, Konkani, Malayalam, Marathi, Nepali, Odiya, Punjabi, Sanskrit, Tamil, and Telugu in a single repository. Relatively newer approaches that emphasiz"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2112.13800","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2021-12-27T17:31:04Z","cross_cats_sorted":[],"title_canon_sha256":"69a6cfb994570b8e77d69185f3fbfe1b2fd078a9b88343c06deae078f2d3f5b4","abstract_canon_sha256":"27eb4872ca3123b8697a65dc2e4ec9889a68cc1c76bb85e829db58e0a36fc4d1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:43:57.100752Z","signature_b64":"tfOzG4vP3fOKCGbRvHKlEUOgQHSRxA3EhNxFzF32v2J8MJySaBPG3S2w4Zx3uYu4XzjTpXcoueqHru0tCM63Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c2cc5aa259e9469f5b9fcc6da3274e3fb3df0dcf9fb3f9844f04afb7aa899681","last_reissued_at":"2026-07-05T03:43:57.100287Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:43:57.100287Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"\"A Passage to India\": Pre-trained Word Embeddings for Indian Languages","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Diptesh Kanojia, Kumar Saunack, Kumar Saurav, Pushpak Bhattacharyya","submitted_at":"2021-12-27T17:31:04Z","abstract_excerpt":"Dense word vectors or 'word embeddings' which encode semantic properties of words, have now become integral to NLP tasks like Machine Translation (MT), Question Answering (QA), Word Sense Disambiguation (WSD), and Information Retrieval (IR). In this paper, we use various existing approaches to create multiple word embeddings for 14 Indian languages. We place these embeddings for all these languages, viz., Assamese, Bengali, Gujarati, Hindi, Kannada, Konkani, Malayalam, Marathi, Nepali, Odiya, Punjabi, Sanskrit, Tamil, and Telugu in a single repository. Relatively newer approaches that emphasiz"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2112.13800","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2112.13800/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2112.13800","created_at":"2026-07-05T03:43:57.100363+00:00"},{"alias_kind":"arxiv_version","alias_value":"2112.13800v1","created_at":"2026-07-05T03:43:57.100363+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2112.13800","created_at":"2026-07-05T03:43:57.100363+00:00"},{"alias_kind":"pith_short_12","alias_value":"YLGFVISZ5FDJ","created_at":"2026-07-05T03:43:57.100363+00:00"},{"alias_kind":"pith_short_16","alias_value":"YLGFVISZ5FDJ6W47","created_at":"2026-07-05T03:43:57.100363+00:00"},{"alias_kind":"pith_short_8","alias_value":"YLGFVISZ","created_at":"2026-07-05T03:43:57.100363+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YLGFVISZ5FDJ6W47ZRW2GJ2OH6","json":"https://pith.science/pith/YLGFVISZ5FDJ6W47ZRW2GJ2OH6.json","graph_json":"https://pith.science/api/pith-number/YLGFVISZ5FDJ6W47ZRW2GJ2OH6/graph.json","events_json":"https://pith.science/api/pith-number/YLGFVISZ5FDJ6W47ZRW2GJ2OH6/events.json","paper":"https://pith.science/paper/YLGFVISZ"},"agent_actions":{"view_html":"https://pith.science/pith/YLGFVISZ5FDJ6W47ZRW2GJ2OH6","download_json":"https://pith.science/pith/YLGFVISZ5FDJ6W47ZRW2GJ2OH6.json","view_paper":"https://pith.science/paper/YLGFVISZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2112.13800&json=true","fetch_graph":"https://pith.science/api/pith-number/YLGFVISZ5FDJ6W47ZRW2GJ2OH6/graph.json","fetch_events":"https://pith.science/api/pith-number/YLGFVISZ5FDJ6W47ZRW2GJ2OH6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YLGFVISZ5FDJ6W47ZRW2GJ2OH6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YLGFVISZ5FDJ6W47ZRW2GJ2OH6/action/storage_attestation","attest_author":"https://pith.science/pith/YLGFVISZ5FDJ6W47ZRW2GJ2OH6/action/author_attestation","sign_citation":"https://pith.science/pith/YLGFVISZ5FDJ6W47ZRW2GJ2OH6/action/citation_signature","submit_replication":"https://pith.science/pith/YLGFVISZ5FDJ6W47ZRW2GJ2OH6/action/replication_record"}},"created_at":"2026-07-05T03:43:57.100363+00:00","updated_at":"2026-07-05T03:43:57.100363+00:00"}