{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:VLAP3GB3PGG4I4H7TKLU4ID36U","short_pith_number":"pith:VLAP3GB3","schema_version":"1.0","canonical_sha256":"aac0fd983b798dc470ff9a974e207bf52f89b1754f177d0641593a3adc375636","source":{"kind":"arxiv","id":"2506.21182","version":1},"attestation_state":"computed","paper":{"title":"Maintaining MTEB: Towards Long Term Usability and Reproducibility of Embedding Benchmarks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.SE"],"primary_cat":"cs.CL","authors_text":"Imene Kerboua, Isaac Chung, Kenneth Enevoldsen, Marton Kardos, Roman Solomatin","submitted_at":"2025-06-26T12:40:48Z","abstract_excerpt":"The Massive Text Embedding Benchmark (MTEB) has become a standard evaluation platform for text embedding models. While previous work has established the core benchmark methodology, this paper focuses on the engineering aspects that ensure MTEB's continued reproducibility and extensibility. We present our approach to maintaining robust continuous integration pipelines that validate dataset integrity, automate test execution, and assess benchmark results' generalizability. We detail the design choices that collectively enhance reproducibility and usability. Furthermore, we discuss our strategies"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.21182","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-26T12:40:48Z","cross_cats_sorted":["cs.AI","cs.SE"],"title_canon_sha256":"feba7ff61f14bb7e461455ab8bc6cc667122f63cb7851bc9e533a03b47eacdc4","abstract_canon_sha256":"441624673519a59ce0f99f48ba11527cde52ba52241148235699886ca46b7775"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:27:41.163500Z","signature_b64":"JdaviqcdOHPzecIyIyTI3GoVM/dzqR/t6ut38GKcJMXmgelDAHPw3lRD6dCTgXubwsKS9I727l1s8xxPy6YyAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"aac0fd983b798dc470ff9a974e207bf52f89b1754f177d0641593a3adc375636","last_reissued_at":"2026-07-05T11:27:41.163050Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:27:41.163050Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Maintaining MTEB: Towards Long Term Usability and Reproducibility of Embedding Benchmarks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.SE"],"primary_cat":"cs.CL","authors_text":"Imene Kerboua, Isaac Chung, Kenneth Enevoldsen, Marton Kardos, Roman Solomatin","submitted_at":"2025-06-26T12:40:48Z","abstract_excerpt":"The Massive Text Embedding Benchmark (MTEB) has become a standard evaluation platform for text embedding models. While previous work has established the core benchmark methodology, this paper focuses on the engineering aspects that ensure MTEB's continued reproducibility and extensibility. We present our approach to maintaining robust continuous integration pipelines that validate dataset integrity, automate test execution, and assess benchmark results' generalizability. We detail the design choices that collectively enhance reproducibility and usability. Furthermore, we discuss our strategies"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.21182","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.21182/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.21182","created_at":"2026-07-05T11:27:41.163108+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.21182v1","created_at":"2026-07-05T11:27:41.163108+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.21182","created_at":"2026-07-05T11:27:41.163108+00:00"},{"alias_kind":"pith_short_12","alias_value":"VLAP3GB3PGG4","created_at":"2026-07-05T11:27:41.163108+00:00"},{"alias_kind":"pith_short_16","alias_value":"VLAP3GB3PGG4I4H7","created_at":"2026-07-05T11:27:41.163108+00:00"},{"alias_kind":"pith_short_8","alias_value":"VLAP3GB3","created_at":"2026-07-05T11:27:41.163108+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.31142","citing_title":"On the Robustness of Multilingual Text Embedding Rankings Across Learning Tasks, Languages, and Benchmark Datasets","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23628","citing_title":"How Hard is it to Rig a Benchmark? A Social Choice Analysis of Leaderboard Robustness","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08421","citing_title":"Beyond Bag-of-Patches: Learning Global Layout via Textual Supervision for Late-Interaction Visual Document Retrieval","ref_index":37,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VLAP3GB3PGG4I4H7TKLU4ID36U","json":"https://pith.science/pith/VLAP3GB3PGG4I4H7TKLU4ID36U.json","graph_json":"https://pith.science/api/pith-number/VLAP3GB3PGG4I4H7TKLU4ID36U/graph.json","events_json":"https://pith.science/api/pith-number/VLAP3GB3PGG4I4H7TKLU4ID36U/events.json","paper":"https://pith.science/paper/VLAP3GB3"},"agent_actions":{"view_html":"https://pith.science/pith/VLAP3GB3PGG4I4H7TKLU4ID36U","download_json":"https://pith.science/pith/VLAP3GB3PGG4I4H7TKLU4ID36U.json","view_paper":"https://pith.science/paper/VLAP3GB3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.21182&json=true","fetch_graph":"https://pith.science/api/pith-number/VLAP3GB3PGG4I4H7TKLU4ID36U/graph.json","fetch_events":"https://pith.science/api/pith-number/VLAP3GB3PGG4I4H7TKLU4ID36U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VLAP3GB3PGG4I4H7TKLU4ID36U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VLAP3GB3PGG4I4H7TKLU4ID36U/action/storage_attestation","attest_author":"https://pith.science/pith/VLAP3GB3PGG4I4H7TKLU4ID36U/action/author_attestation","sign_citation":"https://pith.science/pith/VLAP3GB3PGG4I4H7TKLU4ID36U/action/citation_signature","submit_replication":"https://pith.science/pith/VLAP3GB3PGG4I4H7TKLU4ID36U/action/replication_record"}},"created_at":"2026-07-05T11:27:41.163108+00:00","updated_at":"2026-07-05T11:27:41.163108+00:00"}