{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:276B2642XKKMYZQPTYD5TIZOYZ","short_pith_number":"pith:276B2642","schema_version":"1.0","canonical_sha256":"d7fc1d7b9aba94cc660f9e07d9a32ec673add0711930e59345f1df60583e5e7c","source":{"kind":"arxiv","id":"2401.15713","version":3},"attestation_state":"computed","paper":{"title":"Contrastive Learning and Mixture of Experts Enables Precise Vector Embeddings","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Arjun Patel, Bohdan Khomtchouk, Jason P. Gleghorn, Logan Hallee, Rohan Kapur","submitted_at":"2024-01-28T17:34:42Z","abstract_excerpt":"The advancement of transformer neural networks has significantly elevated the capabilities of sentence similarity models, but they still struggle with highly discriminative tasks and may produce sub-optimal representations of important documents like scientific literature. With the increased reliance on retrieval augmentation and search, representing diverse documents as concise and descriptive vectors is crucial. This paper improves upon the vectors embeddings of scientific text by assembling niche datasets using co-citations as a similarity metric, focusing on biomedical domains. We apply a "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.15713","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-01-28T17:34:42Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"2167ef2a3b3bf50f8340b6e300b1271c50f2fce64a327c000dd7271b696758e8","abstract_canon_sha256":"5547eca8cf152db47f605b15a71e9915adf68bbf75a761c0efe70a4b1d13e32c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:50:45.460398Z","signature_b64":"yjCyZe3pLpux4awGHRvsBTtwJpOk4PUCnYMOp7Rovtqc16DyX3PjyAaRDj1+9LrumGuQjGAosJZQKrce9js9DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d7fc1d7b9aba94cc660f9e07d9a32ec673add0711930e59345f1df60583e5e7c","last_reissued_at":"2026-07-05T09:50:45.459952Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:50:45.459952Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Contrastive Learning and Mixture of Experts Enables Precise Vector Embeddings","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Arjun Patel, Bohdan Khomtchouk, Jason P. Gleghorn, Logan Hallee, Rohan Kapur","submitted_at":"2024-01-28T17:34:42Z","abstract_excerpt":"The advancement of transformer neural networks has significantly elevated the capabilities of sentence similarity models, but they still struggle with highly discriminative tasks and may produce sub-optimal representations of important documents like scientific literature. With the increased reliance on retrieval augmentation and search, representing diverse documents as concise and descriptive vectors is crucial. This paper improves upon the vectors embeddings of scientific text by assembling niche datasets using co-citations as a similarity metric, focusing on biomedical domains. We apply a "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.15713","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.15713/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.15713","created_at":"2026-07-05T09:50:45.460020+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.15713v3","created_at":"2026-07-05T09:50:45.460020+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.15713","created_at":"2026-07-05T09:50:45.460020+00:00"},{"alias_kind":"pith_short_12","alias_value":"276B2642XKKM","created_at":"2026-07-05T09:50:45.460020+00:00"},{"alias_kind":"pith_short_16","alias_value":"276B2642XKKMYZQP","created_at":"2026-07-05T09:50:45.460020+00:00"},{"alias_kind":"pith_short_8","alias_value":"276B2642","created_at":"2026-07-05T09:50:45.460020+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.07972","citing_title":"Training Sparse Mixture Of Experts Text Embedding Models","ref_index":6,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/276B2642XKKMYZQPTYD5TIZOYZ","json":"https://pith.science/pith/276B2642XKKMYZQPTYD5TIZOYZ.json","graph_json":"https://pith.science/api/pith-number/276B2642XKKMYZQPTYD5TIZOYZ/graph.json","events_json":"https://pith.science/api/pith-number/276B2642XKKMYZQPTYD5TIZOYZ/events.json","paper":"https://pith.science/paper/276B2642"},"agent_actions":{"view_html":"https://pith.science/pith/276B2642XKKMYZQPTYD5TIZOYZ","download_json":"https://pith.science/pith/276B2642XKKMYZQPTYD5TIZOYZ.json","view_paper":"https://pith.science/paper/276B2642","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.15713&json=true","fetch_graph":"https://pith.science/api/pith-number/276B2642XKKMYZQPTYD5TIZOYZ/graph.json","fetch_events":"https://pith.science/api/pith-number/276B2642XKKMYZQPTYD5TIZOYZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/276B2642XKKMYZQPTYD5TIZOYZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/276B2642XKKMYZQPTYD5TIZOYZ/action/storage_attestation","attest_author":"https://pith.science/pith/276B2642XKKMYZQPTYD5TIZOYZ/action/author_attestation","sign_citation":"https://pith.science/pith/276B2642XKKMYZQPTYD5TIZOYZ/action/citation_signature","submit_replication":"https://pith.science/pith/276B2642XKKMYZQPTYD5TIZOYZ/action/replication_record"}},"created_at":"2026-07-05T09:50:45.460020+00:00","updated_at":"2026-07-05T09:50:45.460020+00:00"}