{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DYCC66IHCK3TNFARX7OEV22CIK","short_pith_number":"pith:DYCC66IH","schema_version":"1.0","canonical_sha256":"1e042f790712b7369411bfdc4aeb4242aab4d29d526965d909af2189a8c01fb3","source":{"kind":"arxiv","id":"2402.16829","version":1},"attestation_state":"computed","paper":{"title":"GISTEmbed: Guided In-sample Selection of Training Negatives for Text Embedding Fine-tuning","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Aivin V. Solatorio","submitted_at":"2024-02-26T18:55:15Z","abstract_excerpt":"Embedding models are integral to AI applications like semantic search, personalized recommendations, and retrieval augmented generation for LLMs, necessitating high-quality training data. However, the limited scalability of manual data curation prompts the need for automated methods to ensure data integrity. Traditional unsupervised triplet mining automates training data generation, crucial for embedding model training, yet inadvertently injects biases and noise, thereby degrading model performance. Addressing this, we introduce GISTEmbed, a novel strategy that enhances in-batch negative selec"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.16829","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-26T18:55:15Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"5a8de37e6044fdd1e5391c3eaacfcec83f4808f6c73f2b24c8e3c2be80e86e05","abstract_canon_sha256":"55ef3d50b9c79d92d43567fde3fd816878d3e56a5f77d591037f770ce01751a1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:49:26.517847Z","signature_b64":"jREUbTkhmwDQ+ZNd+oPI8XKIg9ntfjTWS9rVO8QXRANJqPxm2F/J2fK60vF/ANIAIHBaahg7fttQ0SeogAWkBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1e042f790712b7369411bfdc4aeb4242aab4d29d526965d909af2189a8c01fb3","last_reissued_at":"2026-07-05T07:49:26.517363Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:49:26.517363Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GISTEmbed: Guided In-sample Selection of Training Negatives for Text Embedding Fine-tuning","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Aivin V. Solatorio","submitted_at":"2024-02-26T18:55:15Z","abstract_excerpt":"Embedding models are integral to AI applications like semantic search, personalized recommendations, and retrieval augmented generation for LLMs, necessitating high-quality training data. However, the limited scalability of manual data curation prompts the need for automated methods to ensure data integrity. Traditional unsupervised triplet mining automates training data generation, crucial for embedding model training, yet inadvertently injects biases and noise, thereby degrading model performance. Addressing this, we introduce GISTEmbed, a novel strategy that enhances in-batch negative selec"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.16829","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.16829/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.16829","created_at":"2026-07-05T07:49:26.517420+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.16829v1","created_at":"2026-07-05T07:49:26.517420+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.16829","created_at":"2026-07-05T07:49:26.517420+00:00"},{"alias_kind":"pith_short_12","alias_value":"DYCC66IHCK3T","created_at":"2026-07-05T07:49:26.517420+00:00"},{"alias_kind":"pith_short_16","alias_value":"DYCC66IHCK3TNFAR","created_at":"2026-07-05T07:49:26.517420+00:00"},{"alias_kind":"pith_short_8","alias_value":"DYCC66IH","created_at":"2026-07-05T07:49:26.517420+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.31692","citing_title":"Overview of the TalentCLEF 2026: Skill and Job Title Intelligence for Human Capital Management","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00084","citing_title":"SentimentLens: Reconciling Sentiment and Ratings via Dual-Modality in the Hospitality Sector","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30473","citing_title":"Field Order Should Not Matter: Permutation-Invariant Embedding Model Fine-Tuning for Structured Metadata Retrieval","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01400","citing_title":"Consistent and Distinctive: LLM Benchmark Efficiency via Maximum Independent Set Prompt Selection on Similarity Graphs","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2507.08480","citing_title":"Improving Korean-English Cross-Lingual Retrieval: A Data-Centric Study of Language Composition and Model Merging","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DYCC66IHCK3TNFARX7OEV22CIK","json":"https://pith.science/pith/DYCC66IHCK3TNFARX7OEV22CIK.json","graph_json":"https://pith.science/api/pith-number/DYCC66IHCK3TNFARX7OEV22CIK/graph.json","events_json":"https://pith.science/api/pith-number/DYCC66IHCK3TNFARX7OEV22CIK/events.json","paper":"https://pith.science/paper/DYCC66IH"},"agent_actions":{"view_html":"https://pith.science/pith/DYCC66IHCK3TNFARX7OEV22CIK","download_json":"https://pith.science/pith/DYCC66IHCK3TNFARX7OEV22CIK.json","view_paper":"https://pith.science/paper/DYCC66IH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.16829&json=true","fetch_graph":"https://pith.science/api/pith-number/DYCC66IHCK3TNFARX7OEV22CIK/graph.json","fetch_events":"https://pith.science/api/pith-number/DYCC66IHCK3TNFARX7OEV22CIK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DYCC66IHCK3TNFARX7OEV22CIK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DYCC66IHCK3TNFARX7OEV22CIK/action/storage_attestation","attest_author":"https://pith.science/pith/DYCC66IHCK3TNFARX7OEV22CIK/action/author_attestation","sign_citation":"https://pith.science/pith/DYCC66IHCK3TNFARX7OEV22CIK/action/citation_signature","submit_replication":"https://pith.science/pith/DYCC66IHCK3TNFARX7OEV22CIK/action/replication_record"}},"created_at":"2026-07-05T07:49:26.517420+00:00","updated_at":"2026-07-05T07:49:26.517420+00:00"}