{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:NZXYC7PV2L4ORM4I6QHERNODLW","short_pith_number":"pith:NZXYC7PV","schema_version":"1.0","canonical_sha256":"6e6f817df5d2f8e8b388f40e48b5c35d910142b7b4367fd427d736d2536b51ea","source":{"kind":"arxiv","id":"2404.05961","version":2},"attestation_state":"computed","paper":{"title":"LLM2Vec: Large Language Models Are Secretly Powerful Text Encoders","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Dzmitry Bahdanau, Marius Mosbach, Nicolas Chapados, Parishad BehnamGhader, Siva Reddy, Vaibhav Adlakha","submitted_at":"2024-04-09T02:51:05Z","abstract_excerpt":"Large decoder-only language models (LLMs) are the state-of-the-art models on most of today's NLP tasks and benchmarks. Yet, the community is only slowly adopting these models for text embedding tasks, which require rich contextualized representations. In this work, we introduce LLM2Vec, a simple unsupervised approach that can transform any decoder-only LLM into a strong text encoder. LLM2Vec consists of three simple steps: 1) enabling bidirectional attention, 2) masked next token prediction, and 3) unsupervised contrastive learning. We demonstrate the effectiveness of LLM2Vec by applying it to"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.05961","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-04-09T02:51:05Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"58ac51054147aa345d4982f892f462ed01824e8c352582229ff821306db96557","abstract_canon_sha256":"d8c323ffe0231c2b496675ad8e1b867d34fae9e11d7f616517b07f609bffb9c8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:57:55.952939Z","signature_b64":"BaZHOmSnDrxActK3Zg7NUf3aGadFnr68Y97j1dXCsciEnc87j4X8xmoDR77tkJ6Yjw45ewcALNnexr6vssg8Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6e6f817df5d2f8e8b388f40e48b5c35d910142b7b4367fd427d736d2536b51ea","last_reissued_at":"2026-07-05T08:57:55.951932Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:57:55.951932Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLM2Vec: Large Language Models Are Secretly Powerful Text Encoders","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Dzmitry Bahdanau, Marius Mosbach, Nicolas Chapados, Parishad BehnamGhader, Siva Reddy, Vaibhav Adlakha","submitted_at":"2024-04-09T02:51:05Z","abstract_excerpt":"Large decoder-only language models (LLMs) are the state-of-the-art models on most of today's NLP tasks and benchmarks. Yet, the community is only slowly adopting these models for text embedding tasks, which require rich contextualized representations. In this work, we introduce LLM2Vec, a simple unsupervised approach that can transform any decoder-only LLM into a strong text encoder. LLM2Vec consists of three simple steps: 1) enabling bidirectional attention, 2) masked next token prediction, and 3) unsupervised contrastive learning. We demonstrate the effectiveness of LLM2Vec by applying it to"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.05961","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.05961/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.05961","created_at":"2026-07-05T08:57:55.952379+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.05961v2","created_at":"2026-07-05T08:57:55.952379+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.05961","created_at":"2026-07-05T08:57:55.952379+00:00"},{"alias_kind":"pith_short_12","alias_value":"NZXYC7PV2L4O","created_at":"2026-07-05T08:57:55.952379+00:00"},{"alias_kind":"pith_short_16","alias_value":"NZXYC7PV2L4ORM4I","created_at":"2026-07-05T08:57:55.952379+00:00"},{"alias_kind":"pith_short_8","alias_value":"NZXYC7PV","created_at":"2026-07-05T08:57:55.952379+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":29,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23791","citing_title":"One Generator, Any Process: LLM-Conditioning for the LHC","ref_index":240,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18781","citing_title":"Lost in a Single Vector: Improving Long-Document Retrieval with Chunk Evidence Aggregation","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11898","citing_title":"GraspLLM: Towards Zero-Shot Generalization on Text-Attributed Graphs with LLMs","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07502","citing_title":"Your UnEmbedding Matrix is Secretly a Feature Lens for Text Embeddings","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04591","citing_title":"Fine-grained Fragment Retrieval in Multi-modal Long-form Dialogues","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31142","citing_title":"On the Robustness of Multilingual Text Embedding Rankings Across Learning Tasks, Languages, and Benchmark Datasets","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23791","citing_title":"One Generator, Any Process: LLM-Conditioning for the LHC","ref_index":244,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26641","citing_title":"OmniRetriever: Any-to-Any Audio-Video-Text Retrieval via Fusion-as-Teacher Distillation","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23312","citing_title":"Towards Generalizable and Efficient Large-Scale Generative Recommenders","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23572","citing_title":"HARNESS-LM: A Three-Phase Training Recipe for Harnessing SLMs in Sponsored Search Retrieval","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2410.22240","citing_title":"Are Decoder-Only Large Language Models the Silver Bullet for Code Search?","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2410.16431","citing_title":"Conjuring Semantic Similarity","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16639","citing_title":"MedMIX: Modality-Internal Expert Fusion for Multimodal Medical Diagnosis","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2412.13663","citing_title":"Smarter, Better, Faster, Longer: A Modern Bidirectional Encoder for Fast, Memory Efficient, and Long Context Finetuning and Inference","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2507.07847","citing_title":"From Ambiguity to Accuracy: The Transformative Effect of Coreference Resolution on Retrieval-Augmented Generation systems","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2507.04590","citing_title":"VLM2Vec-V2: Advancing Multimodal Embedding for Videos, Images, and Visual Documents","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2410.05160","citing_title":"VLM2Vec: Training Vision-Language Models for Massive Multimodal Embedding Tasks","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14448","citing_title":"Think When Needed: Adaptive Reasoning-Driven Multimodal Embeddings with a Dual-LoRA Architecture","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2405.17428","citing_title":"NV-Embed: Improved Techniques for Training LLMs as Generalist Embedding Models","ref_index":125,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04106","citing_title":"InsTraj: Instructing Diffusion Models with Travel Intentions to Generate Real-world Trajectories","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04875","citing_title":"Anticipating Innovation Using Large Language Models","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01097","citing_title":"Interpretable Difficulty-Aware Knowledge Tracing in Tutor-Student Dialogues","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18146","citing_title":"Modular Representation Compression: Adapting LLMs for Efficient and Effective Recommendations","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17866","citing_title":"Latent Abstraction for Retrieval-Augmented Generation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10721","citing_title":"Turning Generators into Retrievers: Unlocking MLLMs for Natural Language-Guided Geo-Localization","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NZXYC7PV2L4ORM4I6QHERNODLW","json":"https://pith.science/pith/NZXYC7PV2L4ORM4I6QHERNODLW.json","graph_json":"https://pith.science/api/pith-number/NZXYC7PV2L4ORM4I6QHERNODLW/graph.json","events_json":"https://pith.science/api/pith-number/NZXYC7PV2L4ORM4I6QHERNODLW/events.json","paper":"https://pith.science/paper/NZXYC7PV"},"agent_actions":{"view_html":"https://pith.science/pith/NZXYC7PV2L4ORM4I6QHERNODLW","download_json":"https://pith.science/pith/NZXYC7PV2L4ORM4I6QHERNODLW.json","view_paper":"https://pith.science/paper/NZXYC7PV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.05961&json=true","fetch_graph":"https://pith.science/api/pith-number/NZXYC7PV2L4ORM4I6QHERNODLW/graph.json","fetch_events":"https://pith.science/api/pith-number/NZXYC7PV2L4ORM4I6QHERNODLW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NZXYC7PV2L4ORM4I6QHERNODLW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NZXYC7PV2L4ORM4I6QHERNODLW/action/storage_attestation","attest_author":"https://pith.science/pith/NZXYC7PV2L4ORM4I6QHERNODLW/action/author_attestation","sign_citation":"https://pith.science/pith/NZXYC7PV2L4ORM4I6QHERNODLW/action/citation_signature","submit_replication":"https://pith.science/pith/NZXYC7PV2L4ORM4I6QHERNODLW/action/replication_record"}},"created_at":"2026-07-05T08:57:55.952379+00:00","updated_at":"2026-07-05T08:57:55.952379+00:00"}