{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:Z5EGW5YI2XMDUVNWNJ57IXLQSV","short_pith_number":"pith:Z5EGW5YI","schema_version":"1.0","canonical_sha256":"cf486b7708d5d83a55b66a7bf45d70954525110f3c15fe56b6b6fac32ebf2fbb","source":{"kind":"arxiv","id":"2104.08027","version":2},"attestation_state":"computed","paper":{"title":"Fast, Effective, and Self-Supervised: Transforming Masked Language Models into Universal Lexical and Sentence Encoders","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Anna Korhonen, Fangyu Liu, Ivan Vuli\\'c, Nigel Collier","submitted_at":"2021-04-16T10:49:56Z","abstract_excerpt":"Pretrained Masked Language Models (MLMs) have revolutionised NLP in recent years. However, previous work has indicated that off-the-shelf MLMs are not effective as universal lexical or sentence encoders without further task-specific fine-tuning on NLI, sentence similarity, or paraphrasing tasks using annotated task data. In this work, we demonstrate that it is possible to turn MLMs into effective universal lexical and sentence encoders even without any additional data and without any supervision. We propose an extremely simple, fast and effective contrastive learning technique, termed Mirror-B"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2104.08027","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2021-04-16T10:49:56Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"9c7488f6eb7e99452d9a58fc48fb3b9c5a1a7f601c41d2febf6e69f41b9b27a7","abstract_canon_sha256":"a47004d923d4bbea2bc137ea48055b66dba33a47cfb6733e02506ec0e20174b5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:12:48.150670Z","signature_b64":"YBgpRMlEjK7kxJL1BqweO9BQxzZt25hwueID6e51H0wm6bbny6JnVOt4ZgdCBO0zlhN3P1us8gvMTsyHjg71Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cf486b7708d5d83a55b66a7bf45d70954525110f3c15fe56b6b6fac32ebf2fbb","last_reissued_at":"2026-07-05T03:12:48.150188Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:12:48.150188Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Fast, Effective, and Self-Supervised: Transforming Masked Language Models into Universal Lexical and Sentence Encoders","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Anna Korhonen, Fangyu Liu, Ivan Vuli\\'c, Nigel Collier","submitted_at":"2021-04-16T10:49:56Z","abstract_excerpt":"Pretrained Masked Language Models (MLMs) have revolutionised NLP in recent years. However, previous work has indicated that off-the-shelf MLMs are not effective as universal lexical or sentence encoders without further task-specific fine-tuning on NLI, sentence similarity, or paraphrasing tasks using annotated task data. In this work, we demonstrate that it is possible to turn MLMs into effective universal lexical and sentence encoders even without any additional data and without any supervision. We propose an extremely simple, fast and effective contrastive learning technique, termed Mirror-B"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2104.08027","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2104.08027/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2104.08027","created_at":"2026-07-05T03:12:48.150253+00:00"},{"alias_kind":"arxiv_version","alias_value":"2104.08027v2","created_at":"2026-07-05T03:12:48.150253+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2104.08027","created_at":"2026-07-05T03:12:48.150253+00:00"},{"alias_kind":"pith_short_12","alias_value":"Z5EGW5YI2XMD","created_at":"2026-07-05T03:12:48.150253+00:00"},{"alias_kind":"pith_short_16","alias_value":"Z5EGW5YI2XMDUVNW","created_at":"2026-07-05T03:12:48.150253+00:00"},{"alias_kind":"pith_short_8","alias_value":"Z5EGW5YI","created_at":"2026-07-05T03:12:48.150253+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.04875","citing_title":"Anticipating Innovation Using Large Language Models","ref_index":41,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Z5EGW5YI2XMDUVNWNJ57IXLQSV","json":"https://pith.science/pith/Z5EGW5YI2XMDUVNWNJ57IXLQSV.json","graph_json":"https://pith.science/api/pith-number/Z5EGW5YI2XMDUVNWNJ57IXLQSV/graph.json","events_json":"https://pith.science/api/pith-number/Z5EGW5YI2XMDUVNWNJ57IXLQSV/events.json","paper":"https://pith.science/paper/Z5EGW5YI"},"agent_actions":{"view_html":"https://pith.science/pith/Z5EGW5YI2XMDUVNWNJ57IXLQSV","download_json":"https://pith.science/pith/Z5EGW5YI2XMDUVNWNJ57IXLQSV.json","view_paper":"https://pith.science/paper/Z5EGW5YI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2104.08027&json=true","fetch_graph":"https://pith.science/api/pith-number/Z5EGW5YI2XMDUVNWNJ57IXLQSV/graph.json","fetch_events":"https://pith.science/api/pith-number/Z5EGW5YI2XMDUVNWNJ57IXLQSV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Z5EGW5YI2XMDUVNWNJ57IXLQSV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Z5EGW5YI2XMDUVNWNJ57IXLQSV/action/storage_attestation","attest_author":"https://pith.science/pith/Z5EGW5YI2XMDUVNWNJ57IXLQSV/action/author_attestation","sign_citation":"https://pith.science/pith/Z5EGW5YI2XMDUVNWNJ57IXLQSV/action/citation_signature","submit_replication":"https://pith.science/pith/Z5EGW5YI2XMDUVNWNJ57IXLQSV/action/replication_record"}},"created_at":"2026-07-05T03:12:48.150253+00:00","updated_at":"2026-07-05T03:12:48.150253+00:00"}