{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:IRGGNXEIJPWZ2IL7LKJNOVTGWX","short_pith_number":"pith:IRGGNXEI","schema_version":"1.0","canonical_sha256":"444c66dc884bed9d217f5a92d75666b5c20d73865c8f95a34d3febfb35b92a67","source":{"kind":"arxiv","id":"2402.15449","version":2},"attestation_state":"computed","paper":{"title":"Repetition Improves Language Model Embeddings","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Aditi Raghunathan, Daniel Fried, Graham Neubig, Jacob Mitchell Springer, Suhas Kotha","submitted_at":"2024-02-23T17:25:10Z","abstract_excerpt":"Bidirectional models are considered essential for strong text embeddings. Recent approaches to adapt autoregressive language models (LMs) into strong text embedding models have largely had the requirement to modify the LM architecture to be bidirectional. We challenge this premise by introducing \"echo embeddings\" which converts autoregressive LMs into high quality text embedding models without changing the architecture or requiring fine-tuning. By repeating the input and extracting embeddings from the repeated tokens -- which have access to all original tokens -- echo embeddings improve over c"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.15449","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-02-23T17:25:10Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"ab22fc389603c7f4884bab415d3272332e75407dd387b091f3af55e412f73cac","abstract_canon_sha256":"3c5f70a033452974dd53d1e14026a72dd7fa3252001641e035fe39283ca3f86b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:05:59.283422Z","signature_b64":"xmP/rQ+IYCBv3s3YSg4ycvXlFTutjZ0o30XXD6VWmARhHxeanGmVyNRtFhsiL3vKCOC+zKYVRARpzinRDNLHDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"444c66dc884bed9d217f5a92d75666b5c20d73865c8f95a34d3febfb35b92a67","last_reissued_at":"2026-07-05T12:05:59.282877Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:05:59.282877Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Repetition Improves Language Model Embeddings","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Aditi Raghunathan, Daniel Fried, Graham Neubig, Jacob Mitchell Springer, Suhas Kotha","submitted_at":"2024-02-23T17:25:10Z","abstract_excerpt":"Bidirectional models are considered essential for strong text embeddings. Recent approaches to adapt autoregressive language models (LMs) into strong text embedding models have largely had the requirement to modify the LM architecture to be bidirectional. We challenge this premise by introducing \"echo embeddings\" which converts autoregressive LMs into high quality text embedding models without changing the architecture or requiring fine-tuning. By repeating the input and extracting embeddings from the repeated tokens -- which have access to all original tokens -- echo embeddings improve over c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.15449","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.15449/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.15449","created_at":"2026-07-05T12:05:59.282939+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.15449v2","created_at":"2026-07-05T12:05:59.282939+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.15449","created_at":"2026-07-05T12:05:59.282939+00:00"},{"alias_kind":"pith_short_12","alias_value":"IRGGNXEIJPWZ","created_at":"2026-07-05T12:05:59.282939+00:00"},{"alias_kind":"pith_short_16","alias_value":"IRGGNXEIJPWZ2IL7","created_at":"2026-07-05T12:05:59.282939+00:00"},{"alias_kind":"pith_short_8","alias_value":"IRGGNXEI","created_at":"2026-07-05T12:05:59.282939+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01792","citing_title":"PARTREP: Learning What to Repeat for Decoder-only LLMs","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05858","citing_title":"ReverseEOL: Improving Training-free Text Embeddings via Text Reversal in Decoder-only LLMs","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2410.22240","citing_title":"Are Decoder-Only Large Language Models the Silver Bullet for Code Search?","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2507.00994","citing_title":"Should We Still Pretrain Encoders with Masked Language Modeling?","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2509.24621","citing_title":"FreeRet: MLLMs as Training-Free Retrievers","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2410.05160","citing_title":"VLM2Vec: Training Vision-Language Models for Massive Multimodal Embedding Tasks","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17866","citing_title":"Latent Abstraction for Retrieval-Augmented Generation","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IRGGNXEIJPWZ2IL7LKJNOVTGWX","json":"https://pith.science/pith/IRGGNXEIJPWZ2IL7LKJNOVTGWX.json","graph_json":"https://pith.science/api/pith-number/IRGGNXEIJPWZ2IL7LKJNOVTGWX/graph.json","events_json":"https://pith.science/api/pith-number/IRGGNXEIJPWZ2IL7LKJNOVTGWX/events.json","paper":"https://pith.science/paper/IRGGNXEI"},"agent_actions":{"view_html":"https://pith.science/pith/IRGGNXEIJPWZ2IL7LKJNOVTGWX","download_json":"https://pith.science/pith/IRGGNXEIJPWZ2IL7LKJNOVTGWX.json","view_paper":"https://pith.science/paper/IRGGNXEI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.15449&json=true","fetch_graph":"https://pith.science/api/pith-number/IRGGNXEIJPWZ2IL7LKJNOVTGWX/graph.json","fetch_events":"https://pith.science/api/pith-number/IRGGNXEIJPWZ2IL7LKJNOVTGWX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IRGGNXEIJPWZ2IL7LKJNOVTGWX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IRGGNXEIJPWZ2IL7LKJNOVTGWX/action/storage_attestation","attest_author":"https://pith.science/pith/IRGGNXEIJPWZ2IL7LKJNOVTGWX/action/author_attestation","sign_citation":"https://pith.science/pith/IRGGNXEIJPWZ2IL7LKJNOVTGWX/action/citation_signature","submit_replication":"https://pith.science/pith/IRGGNXEIJPWZ2IL7LKJNOVTGWX/action/replication_record"}},"created_at":"2026-07-05T12:05:59.282939+00:00","updated_at":"2026-07-05T12:05:59.282939+00:00"}