{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:ZLJIVJZIQT5UDBUUIXMMCETSWV","short_pith_number":"pith:ZLJIVJZI","schema_version":"1.0","canonical_sha256":"cad28aa72884fb41869445d8c11272b57c1522383e33193fbbd6e5efe650db3c","source":{"kind":"arxiv","id":"2608.12875","version":1},"attestation_state":"computed","paper":{"title":"The Embedder's Dilemma: LLMs Are Better, but at What Cost?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Adnan El Assadi, Jinhyuk Lee, Niklas Muennighoff","submitted_at":"2026-08-13T06:39:45Z","abstract_excerpt":"Should you replace your text-embedding pipeline with a large language model? We answer this with a controlled, cost-aware comparison of ten LLMs across six families and 26 embedding models (118M to 14B parameters) on 37 tasks spanning classification, semantic textual similarity (STS), clustering, pair classification, and retrieval. In aggregate the two paradigms are effectively tied: the best LLM (Gemini 3.1 Pro, 77.6) and the best embedding model (77.2) differ by 0.4 points. Their strengths differ by task: LLMs lead on reasoning-heavy retrieval, embedding models lead on classification, and th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.12875","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2026-08-13T06:39:45Z","cross_cats_sorted":[],"title_canon_sha256":"a3f8dfc634e354eed7cf42ad44d8c7f95266d87ddeae97351e6322d048f40881","abstract_canon_sha256":"fdc1ef5aabb8c6a9f58958eea4f5ef398d97b2a9c88c607756c5e67e2bbfea76"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-14T00:49:31.065677Z","signature_b64":"GNjxplM3Uj+wAjYFEzZ0pRwd4qoH/pD2YoaJZ4A+3Lc7oHkha0pXhNUlamZxCxKX/1Arf9qN4Iegj9ojsS8HDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cad28aa72884fb41869445d8c11272b57c1522383e33193fbbd6e5efe650db3c","last_reissued_at":"2026-08-14T00:49:31.053348Z","signature_status":"signed_v1","first_computed_at":"2026-08-14T00:49:31.053348Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Embedder's Dilemma: LLMs Are Better, but at What Cost?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Adnan El Assadi, Jinhyuk Lee, Niklas Muennighoff","submitted_at":"2026-08-13T06:39:45Z","abstract_excerpt":"Should you replace your text-embedding pipeline with a large language model? We answer this with a controlled, cost-aware comparison of ten LLMs across six families and 26 embedding models (118M to 14B parameters) on 37 tasks spanning classification, semantic textual similarity (STS), clustering, pair classification, and retrieval. In aggregate the two paradigms are effectively tied: the best LLM (Gemini 3.1 Pro, 77.6) and the best embedding model (77.2) differ by 0.4 points. Their strengths differ by task: LLMs lead on reasoning-heavy retrieval, embedding models lead on classification, and th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.12875","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.12875/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.12875","created_at":"2026-08-14T00:49:31.054757+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.12875v1","created_at":"2026-08-14T00:49:31.054757+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.12875","created_at":"2026-08-14T00:49:31.054757+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZLJIVJZIQT5U","created_at":"2026-08-14T00:49:31.054757+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZLJIVJZIQT5UDBUU","created_at":"2026-08-14T00:49:31.054757+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZLJIVJZI","created_at":"2026-08-14T00:49:31.054757+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZLJIVJZIQT5UDBUUIXMMCETSWV","json":"https://pith.science/pith/ZLJIVJZIQT5UDBUUIXMMCETSWV.json","graph_json":"https://pith.science/api/pith-number/ZLJIVJZIQT5UDBUUIXMMCETSWV/graph.json","events_json":"https://pith.science/api/pith-number/ZLJIVJZIQT5UDBUUIXMMCETSWV/events.json","paper":"https://pith.science/paper/ZLJIVJZI"},"agent_actions":{"view_html":"https://pith.science/pith/ZLJIVJZIQT5UDBUUIXMMCETSWV","download_json":"https://pith.science/pith/ZLJIVJZIQT5UDBUUIXMMCETSWV.json","view_paper":"https://pith.science/paper/ZLJIVJZI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.12875&json=true","fetch_graph":"https://pith.science/api/pith-number/ZLJIVJZIQT5UDBUUIXMMCETSWV/graph.json","fetch_events":"https://pith.science/api/pith-number/ZLJIVJZIQT5UDBUUIXMMCETSWV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZLJIVJZIQT5UDBUUIXMMCETSWV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZLJIVJZIQT5UDBUUIXMMCETSWV/action/storage_attestation","attest_author":"https://pith.science/pith/ZLJIVJZIQT5UDBUUIXMMCETSWV/action/author_attestation","sign_citation":"https://pith.science/pith/ZLJIVJZIQT5UDBUUIXMMCETSWV/action/citation_signature","submit_replication":"https://pith.science/pith/ZLJIVJZIQT5UDBUUIXMMCETSWV/action/replication_record"}},"created_at":"2026-08-14T00:49:31.054757+00:00","updated_at":"2026-08-14T00:49:31.054757+00:00"}