{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:HDDMEYW6PIK5U7OAHMU7EK7QCT","short_pith_number":"pith:HDDMEYW6","schema_version":"1.0","canonical_sha256":"38c6c262de7a15da7dc03b29f22bf014e858b466370d7dcdaff0ae28a899c7db","source":{"kind":"arxiv","id":"2004.09813","version":2},"attestation_state":"computed","paper":{"title":"Making Monolingual Sentence Embeddings Multilingual using Knowledge Distillation","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Iryna Gurevych, Nils Reimers","submitted_at":"2020-04-21T08:20:25Z","abstract_excerpt":"We present an easy and efficient method to extend existing sentence embedding models to new languages. This allows to create multilingual versions from previously monolingual models. The training is based on the idea that a translated sentence should be mapped to the same location in the vector space as the original sentence. We use the original (monolingual) model to generate sentence embeddings for the source language and then train a new system on translated sentences to mimic the original model. Compared to other methods for training multilingual sentence embeddings, this approach has seve"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2004.09813","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2020-04-21T08:20:25Z","cross_cats_sorted":[],"title_canon_sha256":"ab4a40625f351317b7812801b7d8c23fcc5ffcaed8bc9034810dc8d3a15ce4a6","abstract_canon_sha256":"06dd7d788e313b6e0492373720b6c6c44492af01e0e01e6f125d1eca66eaf9a1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:39:57.514393Z","signature_b64":"kv3+pwo9DWwciQwUyTvZddbdglUUO2ryZVswe3fEw6NYVK3D2ngIgAkI05NWznikKeJOs/nui5mGjJrVXtIKAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"38c6c262de7a15da7dc03b29f22bf014e858b466370d7dcdaff0ae28a899c7db","last_reissued_at":"2026-07-05T01:39:57.513897Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:39:57.513897Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Making Monolingual Sentence Embeddings Multilingual using Knowledge Distillation","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Iryna Gurevych, Nils Reimers","submitted_at":"2020-04-21T08:20:25Z","abstract_excerpt":"We present an easy and efficient method to extend existing sentence embedding models to new languages. This allows to create multilingual versions from previously monolingual models. The training is based on the idea that a translated sentence should be mapped to the same location in the vector space as the original sentence. We use the original (monolingual) model to generate sentence embeddings for the source language and then train a new system on translated sentences to mimic the original model. Compared to other methods for training multilingual sentence embeddings, this approach has seve"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2004.09813","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2004.09813/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2004.09813","created_at":"2026-07-05T01:39:57.513958+00:00"},{"alias_kind":"arxiv_version","alias_value":"2004.09813v2","created_at":"2026-07-05T01:39:57.513958+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2004.09813","created_at":"2026-07-05T01:39:57.513958+00:00"},{"alias_kind":"pith_short_12","alias_value":"HDDMEYW6PIK5","created_at":"2026-07-05T01:39:57.513958+00:00"},{"alias_kind":"pith_short_16","alias_value":"HDDMEYW6PIK5U7OA","created_at":"2026-07-05T01:39:57.513958+00:00"},{"alias_kind":"pith_short_8","alias_value":"HDDMEYW6","created_at":"2026-07-05T01:39:57.513958+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03307","citing_title":"Generalizing Graph Foundation Models via Hyperbolic Retrieval-Augmented Generation","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28531","citing_title":"A Good Talk Does not Look Like a Summary, It Teaches You! Measuring Takeaways from Paper-to-Video Talks","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26595","citing_title":"Cordyceps: Covert Control Attacks on LLMs via Data Poisoning","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28190","citing_title":"The Harder Text Embedding Benchmark (HTEB): Beyond One-dimensional Static Robustness","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29992","citing_title":"Adapting Multilingual Embedding Models to Turkish via Cross-Lingual Tokenizer Surgery and Offline Distillation","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23753","citing_title":"SeedER: Seed-and-Expand Retrieval from Knowledge Graphs","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2406.11354","citing_title":"Preserving Knowledge in Large Language Model with Model-Agnostic Self-Decompression","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26526","citing_title":"Identifying and Characterizing Semantic Clones of Solidity Functions","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18525","citing_title":"Towards Better Static Code Analysis Reports: Sentence Transformer-based Filtering of Non-Actionable Alerts","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2410.02713","citing_title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","ref_index":43,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HDDMEYW6PIK5U7OAHMU7EK7QCT","json":"https://pith.science/pith/HDDMEYW6PIK5U7OAHMU7EK7QCT.json","graph_json":"https://pith.science/api/pith-number/HDDMEYW6PIK5U7OAHMU7EK7QCT/graph.json","events_json":"https://pith.science/api/pith-number/HDDMEYW6PIK5U7OAHMU7EK7QCT/events.json","paper":"https://pith.science/paper/HDDMEYW6"},"agent_actions":{"view_html":"https://pith.science/pith/HDDMEYW6PIK5U7OAHMU7EK7QCT","download_json":"https://pith.science/pith/HDDMEYW6PIK5U7OAHMU7EK7QCT.json","view_paper":"https://pith.science/paper/HDDMEYW6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2004.09813&json=true","fetch_graph":"https://pith.science/api/pith-number/HDDMEYW6PIK5U7OAHMU7EK7QCT/graph.json","fetch_events":"https://pith.science/api/pith-number/HDDMEYW6PIK5U7OAHMU7EK7QCT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HDDMEYW6PIK5U7OAHMU7EK7QCT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HDDMEYW6PIK5U7OAHMU7EK7QCT/action/storage_attestation","attest_author":"https://pith.science/pith/HDDMEYW6PIK5U7OAHMU7EK7QCT/action/author_attestation","sign_citation":"https://pith.science/pith/HDDMEYW6PIK5U7OAHMU7EK7QCT/action/citation_signature","submit_replication":"https://pith.science/pith/HDDMEYW6PIK5U7OAHMU7EK7QCT/action/replication_record"}},"created_at":"2026-07-05T01:39:57.513958+00:00","updated_at":"2026-07-05T01:39:57.513958+00:00"}