{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:TZURPOD6MKH3A7HJCFAH4VMHAS","short_pith_number":"pith:TZURPOD6","schema_version":"1.0","canonical_sha256":"9e6917b87e628fb07ce911407e558704b74610b9158920741444471862e150b8","source":{"kind":"arxiv","id":"2109.14796","version":1},"attestation_state":"computed","paper":{"title":"Phonetic Word Embeddings","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Balakrishna Pailla, Kunal Dhawan, Rahul Sharma","submitted_at":"2021-09-30T01:46:01Z","abstract_excerpt":"This work presents a novel methodology for calculating the phonetic similarity between words taking motivation from the human perception of sounds. This metric is employed to learn a continuous vector embedding space that groups similar sounding words together and can be used for various downstream computational phonology tasks. The efficacy of the method is presented for two different languages (English, Hindi) and performance gains over previous reported works are discussed on established tests for predicting phonetic similarity. To address limited benchmarking mechanisms in this field, we a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2109.14796","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2021-09-30T01:46:01Z","cross_cats_sorted":[],"title_canon_sha256":"25de230f294e32a50aa21df04ad9f020bdaff4ed60796e35fff45a9619e428b8","abstract_canon_sha256":"bd84e2fed422202d95cb722e315c608193f092f697844fc3bc9282615eca87d0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:18:44.476269Z","signature_b64":"u2sypbjFOi4cU4Onl3A8FzfAs+Orj+rseHwPta1Mk5pBAgl/1kuOqmrhTIIvKJnTvgIoA4aTCUBP5xJVw11MDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9e6917b87e628fb07ce911407e558704b74610b9158920741444471862e150b8","last_reissued_at":"2026-07-05T03:18:44.475921Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:18:44.475921Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Phonetic Word Embeddings","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Balakrishna Pailla, Kunal Dhawan, Rahul Sharma","submitted_at":"2021-09-30T01:46:01Z","abstract_excerpt":"This work presents a novel methodology for calculating the phonetic similarity between words taking motivation from the human perception of sounds. This metric is employed to learn a continuous vector embedding space that groups similar sounding words together and can be used for various downstream computational phonology tasks. The efficacy of the method is presented for two different languages (English, Hindi) and performance gains over previous reported works are discussed on established tests for predicting phonetic similarity. To address limited benchmarking mechanisms in this field, we a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2109.14796","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2109.14796/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2109.14796","created_at":"2026-07-05T03:18:44.475984+00:00"},{"alias_kind":"arxiv_version","alias_value":"2109.14796v1","created_at":"2026-07-05T03:18:44.475984+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2109.14796","created_at":"2026-07-05T03:18:44.475984+00:00"},{"alias_kind":"pith_short_12","alias_value":"TZURPOD6MKH3","created_at":"2026-07-05T03:18:44.475984+00:00"},{"alias_kind":"pith_short_16","alias_value":"TZURPOD6MKH3A7HJ","created_at":"2026-07-05T03:18:44.475984+00:00"},{"alias_kind":"pith_short_8","alias_value":"TZURPOD6","created_at":"2026-07-05T03:18:44.475984+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.00911","citing_title":"The Cognate Data Bottleneck in Language Phylogenetics","ref_index":30,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TZURPOD6MKH3A7HJCFAH4VMHAS","json":"https://pith.science/pith/TZURPOD6MKH3A7HJCFAH4VMHAS.json","graph_json":"https://pith.science/api/pith-number/TZURPOD6MKH3A7HJCFAH4VMHAS/graph.json","events_json":"https://pith.science/api/pith-number/TZURPOD6MKH3A7HJCFAH4VMHAS/events.json","paper":"https://pith.science/paper/TZURPOD6"},"agent_actions":{"view_html":"https://pith.science/pith/TZURPOD6MKH3A7HJCFAH4VMHAS","download_json":"https://pith.science/pith/TZURPOD6MKH3A7HJCFAH4VMHAS.json","view_paper":"https://pith.science/paper/TZURPOD6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2109.14796&json=true","fetch_graph":"https://pith.science/api/pith-number/TZURPOD6MKH3A7HJCFAH4VMHAS/graph.json","fetch_events":"https://pith.science/api/pith-number/TZURPOD6MKH3A7HJCFAH4VMHAS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TZURPOD6MKH3A7HJCFAH4VMHAS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TZURPOD6MKH3A7HJCFAH4VMHAS/action/storage_attestation","attest_author":"https://pith.science/pith/TZURPOD6MKH3A7HJCFAH4VMHAS/action/author_attestation","sign_citation":"https://pith.science/pith/TZURPOD6MKH3A7HJCFAH4VMHAS/action/citation_signature","submit_replication":"https://pith.science/pith/TZURPOD6MKH3A7HJCFAH4VMHAS/action/replication_record"}},"created_at":"2026-07-05T03:18:44.475984+00:00","updated_at":"2026-07-05T03:18:44.475984+00:00"}