{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:7MQHA7X3HPVJJIUAZ2F7KTXNER","short_pith_number":"pith:7MQHA7X3","schema_version":"1.0","canonical_sha256":"fb20707efb3bea94a280ce8bf54eed2451cb895a9615e245fc5db03694ffd564","source":{"kind":"arxiv","id":"2412.03223","version":1},"attestation_state":"computed","paper":{"title":"Linq-Embed-Mistral Technical Report","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chanyeol Choi, Jihoon Kwon, Junseong Kim, Jy-yong Sohn, MinKyung Cho, Sangmo Gu, Seolhwa Lee, Yejin Kim","submitted_at":"2024-12-04T11:18:32Z","abstract_excerpt":"This report explores the enhancement of text retrieval performance using advanced data refinement techniques. We develop Linq-Embed-Mistral\\footnote{\\url{https://huggingface.co/Linq-AI-Research/Linq-Embed-Mistral}} by building on the E5-mistral and Mistral-7B-v0.1 models, focusing on sophisticated data crafting, data filtering, and negative mining methods, which are highly tailored to each task, applied to both existing benchmark dataset and highly tailored synthetic dataset generated via large language models (LLMs). Linq-Embed-Mistral excels in the MTEB benchmarks (as of May 29, 2024), achie"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.03223","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-12-04T11:18:32Z","cross_cats_sorted":[],"title_canon_sha256":"f7ce2e7fb7356d5f17c5bab44af3242a90b7dbf8b3bd54e4612bf0ac75387fcf","abstract_canon_sha256":"0897d5de2d6f90003bcfd9e5abfa812b00ba9506b5b71f89b1738b313ee32fa6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:44:26.692562Z","signature_b64":"e5YmXyt71iOiTgPLgVedA7pyCtb0YvNxTgwpK6cddtDGVpB4TDSSPd/NNJCEwywe/8zGRlbK0mxVm8o1M1sxAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fb20707efb3bea94a280ce8bf54eed2451cb895a9615e245fc5db03694ffd564","last_reissued_at":"2026-07-05T09:44:26.692094Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:44:26.692094Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Linq-Embed-Mistral Technical Report","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chanyeol Choi, Jihoon Kwon, Junseong Kim, Jy-yong Sohn, MinKyung Cho, Sangmo Gu, Seolhwa Lee, Yejin Kim","submitted_at":"2024-12-04T11:18:32Z","abstract_excerpt":"This report explores the enhancement of text retrieval performance using advanced data refinement techniques. We develop Linq-Embed-Mistral\\footnote{\\url{https://huggingface.co/Linq-AI-Research/Linq-Embed-Mistral}} by building on the E5-mistral and Mistral-7B-v0.1 models, focusing on sophisticated data crafting, data filtering, and negative mining methods, which are highly tailored to each task, applied to both existing benchmark dataset and highly tailored synthetic dataset generated via large language models (LLMs). Linq-Embed-Mistral excels in the MTEB benchmarks (as of May 29, 2024), achie"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.03223","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.03223/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.03223","created_at":"2026-07-05T09:44:26.692149+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.03223v1","created_at":"2026-07-05T09:44:26.692149+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.03223","created_at":"2026-07-05T09:44:26.692149+00:00"},{"alias_kind":"pith_short_12","alias_value":"7MQHA7X3HPVJ","created_at":"2026-07-05T09:44:26.692149+00:00"},{"alias_kind":"pith_short_16","alias_value":"7MQHA7X3HPVJJIUA","created_at":"2026-07-05T09:44:26.692149+00:00"},{"alias_kind":"pith_short_8","alias_value":"7MQHA7X3","created_at":"2026-07-05T09:44:26.692149+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25674","citing_title":"BitNet Text Embeddings","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31142","citing_title":"On the Robustness of Multilingual Text Embedding Rankings Across Learning Tasks, Languages, and Benchmark Datasets","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22247","citing_title":"IdioLink: Retrieving Meaning Beyond Words Across Idiomatic and Literal Expressions","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2507.07847","citing_title":"From Ambiguity to Accuracy: The Transformative Effect of Coreference Resolution on Retrieval-Augmented Generation systems","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2508.03306","citing_title":"Reliable Evaluation Protocol for Low-Precision Retrieval","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2510.05038","citing_title":"Guided Query Refinement: Multimodal Hybrid Retrieval with Test-Time Optimization","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12487","citing_title":"Task-Adaptive Embedding Refinement via Test-time LLM Guidance","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19834","citing_title":"KD-Judge: A Knowledge-Driven Automated Judge Framework for Functional Fitness Movements on Edge Devices","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16576","citing_title":"On the Robustness of LLM-Based Dense Retrievers: A Systematic Analysis of Generalizability and Stability","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18199","citing_title":"Linear-Time and Constant-Memory Text Embeddings Based on Recurrent Language Models","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7MQHA7X3HPVJJIUAZ2F7KTXNER","json":"https://pith.science/pith/7MQHA7X3HPVJJIUAZ2F7KTXNER.json","graph_json":"https://pith.science/api/pith-number/7MQHA7X3HPVJJIUAZ2F7KTXNER/graph.json","events_json":"https://pith.science/api/pith-number/7MQHA7X3HPVJJIUAZ2F7KTXNER/events.json","paper":"https://pith.science/paper/7MQHA7X3"},"agent_actions":{"view_html":"https://pith.science/pith/7MQHA7X3HPVJJIUAZ2F7KTXNER","download_json":"https://pith.science/pith/7MQHA7X3HPVJJIUAZ2F7KTXNER.json","view_paper":"https://pith.science/paper/7MQHA7X3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.03223&json=true","fetch_graph":"https://pith.science/api/pith-number/7MQHA7X3HPVJJIUAZ2F7KTXNER/graph.json","fetch_events":"https://pith.science/api/pith-number/7MQHA7X3HPVJJIUAZ2F7KTXNER/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7MQHA7X3HPVJJIUAZ2F7KTXNER/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7MQHA7X3HPVJJIUAZ2F7KTXNER/action/storage_attestation","attest_author":"https://pith.science/pith/7MQHA7X3HPVJJIUAZ2F7KTXNER/action/author_attestation","sign_citation":"https://pith.science/pith/7MQHA7X3HPVJJIUAZ2F7KTXNER/action/citation_signature","submit_replication":"https://pith.science/pith/7MQHA7X3HPVJJIUAZ2F7KTXNER/action/replication_record"}},"created_at":"2026-07-05T09:44:26.692149+00:00","updated_at":"2026-07-05T09:44:26.692149+00:00"}