{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KEMRMOBNMQYLASAPXQ3CXKZJ4P","short_pith_number":"pith:KEMRMOBN","schema_version":"1.0","canonical_sha256":"511916382d6430b0480fbc362bab29e3fb01af27854b3f297129a54539ce48a7","source":{"kind":"arxiv","id":"2407.08275","version":1},"attestation_state":"computed","paper":{"title":"Beyond Benchmarks: Evaluating Embedding Model Similarity for Retrieval Augmented Generation Systems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.IR","authors_text":"Jelena Mitrovic, Kanishka Ghosh Dastidar, Laura Caspari, Michael Granitzer, Saber Zerhoudi","submitted_at":"2024-07-11T08:24:16Z","abstract_excerpt":"The choice of embedding model is a crucial step in the design of Retrieval Augmented Generation (RAG) systems. Given the sheer volume of available options, identifying clusters of similar models streamlines this model selection process. Relying solely on benchmark performance scores only allows for a weak assessment of model similarity. Thus, in this study, we evaluate the similarity of embedding models within the context of RAG systems. Our assessment is two-fold: We use Centered Kernel Alignment to compare embeddings on a pair-wise level. Additionally, as it is especially pertinent to RAG sy"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.08275","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.IR","submitted_at":"2024-07-11T08:24:16Z","cross_cats_sorted":[],"title_canon_sha256":"4200afe397cc555d8f001cc74aec050ecedaf15ccd0af92614dc682322c11810","abstract_canon_sha256":"8e20279749fd0851656b37373c6fd282550e660441ca73eeeaf13bcb195e4288"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:42:46.024603Z","signature_b64":"pmyM4AP8xE1jsL6qT66k/0KQmKLoVMpMOGOXx/elCnfOF0Dj+zGPT1EFndexHcwTlLsVUkiVJNi5jTyFOTzWDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"511916382d6430b0480fbc362bab29e3fb01af27854b3f297129a54539ce48a7","last_reissued_at":"2026-07-05T08:42:46.024196Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:42:46.024196Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Beyond Benchmarks: Evaluating Embedding Model Similarity for Retrieval Augmented Generation Systems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.IR","authors_text":"Jelena Mitrovic, Kanishka Ghosh Dastidar, Laura Caspari, Michael Granitzer, Saber Zerhoudi","submitted_at":"2024-07-11T08:24:16Z","abstract_excerpt":"The choice of embedding model is a crucial step in the design of Retrieval Augmented Generation (RAG) systems. Given the sheer volume of available options, identifying clusters of similar models streamlines this model selection process. Relying solely on benchmark performance scores only allows for a weak assessment of model similarity. Thus, in this study, we evaluate the similarity of embedding models within the context of RAG systems. Our assessment is two-fold: We use Centered Kernel Alignment to compare embeddings on a pair-wise level. Additionally, as it is especially pertinent to RAG sy"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.08275","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.08275/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.08275","created_at":"2026-07-05T08:42:46.024262+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.08275v1","created_at":"2026-07-05T08:42:46.024262+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.08275","created_at":"2026-07-05T08:42:46.024262+00:00"},{"alias_kind":"pith_short_12","alias_value":"KEMRMOBNMQYL","created_at":"2026-07-05T08:42:46.024262+00:00"},{"alias_kind":"pith_short_16","alias_value":"KEMRMOBNMQYLASAP","created_at":"2026-07-05T08:42:46.024262+00:00"},{"alias_kind":"pith_short_8","alias_value":"KEMRMOBN","created_at":"2026-07-05T08:42:46.024262+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01852","citing_title":"Evaluating Chunking Strategies for Retrieval-Augmented Generation on Academic Texts","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27306","citing_title":"NuggetIndex: Governed Atomic Retrieval for Maintainable RAG","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26653","citing_title":"AgentSim: A Platform for Verifiable Agent-Trace Simulation","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KEMRMOBNMQYLASAPXQ3CXKZJ4P","json":"https://pith.science/pith/KEMRMOBNMQYLASAPXQ3CXKZJ4P.json","graph_json":"https://pith.science/api/pith-number/KEMRMOBNMQYLASAPXQ3CXKZJ4P/graph.json","events_json":"https://pith.science/api/pith-number/KEMRMOBNMQYLASAPXQ3CXKZJ4P/events.json","paper":"https://pith.science/paper/KEMRMOBN"},"agent_actions":{"view_html":"https://pith.science/pith/KEMRMOBNMQYLASAPXQ3CXKZJ4P","download_json":"https://pith.science/pith/KEMRMOBNMQYLASAPXQ3CXKZJ4P.json","view_paper":"https://pith.science/paper/KEMRMOBN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.08275&json=true","fetch_graph":"https://pith.science/api/pith-number/KEMRMOBNMQYLASAPXQ3CXKZJ4P/graph.json","fetch_events":"https://pith.science/api/pith-number/KEMRMOBNMQYLASAPXQ3CXKZJ4P/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KEMRMOBNMQYLASAPXQ3CXKZJ4P/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KEMRMOBNMQYLASAPXQ3CXKZJ4P/action/storage_attestation","attest_author":"https://pith.science/pith/KEMRMOBNMQYLASAPXQ3CXKZJ4P/action/author_attestation","sign_citation":"https://pith.science/pith/KEMRMOBNMQYLASAPXQ3CXKZJ4P/action/citation_signature","submit_replication":"https://pith.science/pith/KEMRMOBNMQYLASAPXQ3CXKZJ4P/action/replication_record"}},"created_at":"2026-07-05T08:42:46.024262+00:00","updated_at":"2026-07-05T08:42:46.024262+00:00"}