{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PQINM2ETNM6UXPTVN6HLBA3DJD","short_pith_number":"pith:PQINM2ET","schema_version":"1.0","canonical_sha256":"7c10d668936b3d4bbe756f8eb0836348d9270388536f0d9631e5d5b31136f7db","source":{"kind":"arxiv","id":"2412.19048","version":2},"attestation_state":"computed","paper":{"title":"Jasper and Stella: distillation of SOTA embedding models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.IR","authors_text":"Dun Zhang, Fulong Wang, Jiacheng Li, Ziyang Zeng","submitted_at":"2024-12-26T04:05:28Z","abstract_excerpt":"A crucial component in many deep learning applications, such as Frequently Asked Questions (FAQ) and Retrieval-Augmented Generation (RAG), is dense retrieval. In this process, embedding models transform raw text into numerical vectors. However, the embedding models that currently excel on text embedding benchmarks, like the Massive Text Embedding Benchmark (MTEB), often have numerous parameters and high vector dimensionality. This poses challenges for their application in real-world scenarios. To address this issue, we propose a novel multi-stage distillation framework that enables a smaller s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.19048","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.IR","submitted_at":"2024-12-26T04:05:28Z","cross_cats_sorted":[],"title_canon_sha256":"b37907e0632a07dea094a778e29255d2d65d7800444e50bdf3e5e8fa86dad68f","abstract_canon_sha256":"b7c9f43cbca83224363fa53fc67a2ee20ef1ac9f25cafef42ac83cd9f6956aba"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:04:21.327274Z","signature_b64":"Z7e/6t6Xa1GGf0/ptimi/YEe4OQzlXodOjj6MHUqh05BGZdCreVuJ1CPc1DmJLyZja7RrPEcGomJWR610RCBCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7c10d668936b3d4bbe756f8eb0836348d9270388536f0d9631e5d5b31136f7db","last_reissued_at":"2026-07-05T10:04:21.326859Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:04:21.326859Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Jasper and Stella: distillation of SOTA embedding models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.IR","authors_text":"Dun Zhang, Fulong Wang, Jiacheng Li, Ziyang Zeng","submitted_at":"2024-12-26T04:05:28Z","abstract_excerpt":"A crucial component in many deep learning applications, such as Frequently Asked Questions (FAQ) and Retrieval-Augmented Generation (RAG), is dense retrieval. In this process, embedding models transform raw text into numerical vectors. However, the embedding models that currently excel on text embedding benchmarks, like the Massive Text Embedding Benchmark (MTEB), often have numerous parameters and high vector dimensionality. This poses challenges for their application in real-world scenarios. To address this issue, we propose a novel multi-stage distillation framework that enables a smaller s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.19048","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.19048/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.19048","created_at":"2026-07-05T10:04:21.326915+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.19048v2","created_at":"2026-07-05T10:04:21.326915+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.19048","created_at":"2026-07-05T10:04:21.326915+00:00"},{"alias_kind":"pith_short_12","alias_value":"PQINM2ETNM6U","created_at":"2026-07-05T10:04:21.326915+00:00"},{"alias_kind":"pith_short_16","alias_value":"PQINM2ETNM6UXPTV","created_at":"2026-07-05T10:04:21.326915+00:00"},{"alias_kind":"pith_short_8","alias_value":"PQINM2ET","created_at":"2026-07-05T10:04:21.326915+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.24297","citing_title":"Benchmarking Patent Embeddings: A Multi-Task Evaluation of 22 Models Across Retrieval, Classification, and Clustering","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27345","citing_title":"MATCHA: Matching Text via Contrastive Semantic Alignment","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28190","citing_title":"The Harder Text Embedding Benchmark (HTEB): Beyond One-dimensional Static Robustness","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22247","citing_title":"IdioLink: Retrieving Meaning Beyond Words Across Idiomatic and Literal Expressions","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2510.04905","citing_title":"Retrieval-Augmented Code Generation: A Survey with Focus on Repository-Level Approaches","ref_index":162,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17823","citing_title":"Why We Look Where We Look: Emergent Human-like Fixations of a Foveated Visual Language Model Maximizing Scene Understanding","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2507.07847","citing_title":"From Ambiguity to Accuracy: The Transformative Effect of Coreference Resolution on Retrieval-Augmented Generation systems","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2507.06419","citing_title":"Teach a Reward Model to Correct Itself: Reward Guided Adversarial Failure Discovery for Robust Reward Modeling","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2510.01801","citing_title":"Detecting LLM-Generated Spam Reviews by Integrating Language Model Embeddings and Graph Neural Network","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16301","citing_title":"Domain-Specific Query Understanding for Automotive Applications: A Modular and Scalable Approach","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2602.12783","citing_title":"SQuTR: A Robustness Benchmark for Spoken Query to Text Retrieval under Acoustic Noise","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2602.15547","citing_title":"jina-embeddings-v5-text: Task-Targeted Embedding Distillation","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13047","citing_title":"Revealing the Gap in Human and VLM Scene Perception through Counterfactual Semantic Saliency","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08360","citing_title":"Embeddings for Preferences, Not Semantics","ref_index":65,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PQINM2ETNM6UXPTVN6HLBA3DJD","json":"https://pith.science/pith/PQINM2ETNM6UXPTVN6HLBA3DJD.json","graph_json":"https://pith.science/api/pith-number/PQINM2ETNM6UXPTVN6HLBA3DJD/graph.json","events_json":"https://pith.science/api/pith-number/PQINM2ETNM6UXPTVN6HLBA3DJD/events.json","paper":"https://pith.science/paper/PQINM2ET"},"agent_actions":{"view_html":"https://pith.science/pith/PQINM2ETNM6UXPTVN6HLBA3DJD","download_json":"https://pith.science/pith/PQINM2ETNM6UXPTVN6HLBA3DJD.json","view_paper":"https://pith.science/paper/PQINM2ET","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.19048&json=true","fetch_graph":"https://pith.science/api/pith-number/PQINM2ETNM6UXPTVN6HLBA3DJD/graph.json","fetch_events":"https://pith.science/api/pith-number/PQINM2ETNM6UXPTVN6HLBA3DJD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PQINM2ETNM6UXPTVN6HLBA3DJD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PQINM2ETNM6UXPTVN6HLBA3DJD/action/storage_attestation","attest_author":"https://pith.science/pith/PQINM2ETNM6UXPTVN6HLBA3DJD/action/author_attestation","sign_citation":"https://pith.science/pith/PQINM2ETNM6UXPTVN6HLBA3DJD/action/citation_signature","submit_replication":"https://pith.science/pith/PQINM2ETNM6UXPTVN6HLBA3DJD/action/replication_record"}},"created_at":"2026-07-05T10:04:21.326915+00:00","updated_at":"2026-07-05T10:04:21.326915+00:00"}