{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:7WRAHT5R7PQCGCSLZ4VZKPD6QL","short_pith_number":"pith:7WRAHT5R","schema_version":"1.0","canonical_sha256":"fda203cfb1fbe0230a4bcf2b953c7e82dfa8f34be7c24f1f581f35a8a7bea790","source":{"kind":"arxiv","id":"2409.18511","version":4},"attestation_state":"computed","paper":{"title":"Do We Need Domain-Specific Embedding Models? An Empirical Investigation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.IR"],"primary_cat":"cs.CL","authors_text":"Yixuan Tang, Yi Yang","submitted_at":"2024-09-27T07:46:06Z","abstract_excerpt":"Embedding models play a crucial role in representing and retrieving information across various NLP applications. Recent advancements in Large Language Models (LLMs) have further enhanced the performance of embedding models, which are trained on massive amounts of text covering almost every domain. These models are often benchmarked on general-purpose datasets like Massive Text Embedding Benchmark (MTEB), where they demonstrate superior performance. However, a critical question arises: Is the development of domain-specific embedding models necessary when general-purpose models are trained on va"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.18511","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-09-27T07:46:06Z","cross_cats_sorted":["cs.IR"],"title_canon_sha256":"74c88a557606b7db643f4b17da0f264607933aff11556bc008eb2436793dd354","abstract_canon_sha256":"cd27cd0b3c4e657af0a20b1403fa7e8e111c80cf70a5ebe2cff10f64b4839045"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:15:45.635825Z","signature_b64":"nTf6dRQX4xXCZ4T9zXJlSDbTz8SdOL8JWMXeQDK7SQqlNpDvSLqT8J7z9hX/TgIKfw9ehq6PlXVUf28ZciAPCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fda203cfb1fbe0230a4bcf2b953c7e82dfa8f34be7c24f1f581f35a8a7bea790","last_reissued_at":"2026-07-05T10:15:45.635338Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:15:45.635338Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Do We Need Domain-Specific Embedding Models? An Empirical Investigation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.IR"],"primary_cat":"cs.CL","authors_text":"Yixuan Tang, Yi Yang","submitted_at":"2024-09-27T07:46:06Z","abstract_excerpt":"Embedding models play a crucial role in representing and retrieving information across various NLP applications. Recent advancements in Large Language Models (LLMs) have further enhanced the performance of embedding models, which are trained on massive amounts of text covering almost every domain. These models are often benchmarked on general-purpose datasets like Massive Text Embedding Benchmark (MTEB), where they demonstrate superior performance. However, a critical question arises: Is the development of domain-specific embedding models necessary when general-purpose models are trained on va"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.18511","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.18511/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.18511","created_at":"2026-07-05T10:15:45.635395+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.18511v4","created_at":"2026-07-05T10:15:45.635395+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.18511","created_at":"2026-07-05T10:15:45.635395+00:00"},{"alias_kind":"pith_short_12","alias_value":"7WRAHT5R7PQC","created_at":"2026-07-05T10:15:45.635395+00:00"},{"alias_kind":"pith_short_16","alias_value":"7WRAHT5R7PQCGCSL","created_at":"2026-07-05T10:15:45.635395+00:00"},{"alias_kind":"pith_short_8","alias_value":"7WRAHT5R","created_at":"2026-07-05T10:15:45.635395+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.06335","citing_title":"FinBERT2: A Specialized Bidirectional Encoder for Bridging the Gap in Finance-Specific Deployment of Large Language Models","ref_index":34,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7WRAHT5R7PQCGCSLZ4VZKPD6QL","json":"https://pith.science/pith/7WRAHT5R7PQCGCSLZ4VZKPD6QL.json","graph_json":"https://pith.science/api/pith-number/7WRAHT5R7PQCGCSLZ4VZKPD6QL/graph.json","events_json":"https://pith.science/api/pith-number/7WRAHT5R7PQCGCSLZ4VZKPD6QL/events.json","paper":"https://pith.science/paper/7WRAHT5R"},"agent_actions":{"view_html":"https://pith.science/pith/7WRAHT5R7PQCGCSLZ4VZKPD6QL","download_json":"https://pith.science/pith/7WRAHT5R7PQCGCSLZ4VZKPD6QL.json","view_paper":"https://pith.science/paper/7WRAHT5R","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.18511&json=true","fetch_graph":"https://pith.science/api/pith-number/7WRAHT5R7PQCGCSLZ4VZKPD6QL/graph.json","fetch_events":"https://pith.science/api/pith-number/7WRAHT5R7PQCGCSLZ4VZKPD6QL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7WRAHT5R7PQCGCSLZ4VZKPD6QL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7WRAHT5R7PQCGCSLZ4VZKPD6QL/action/storage_attestation","attest_author":"https://pith.science/pith/7WRAHT5R7PQCGCSLZ4VZKPD6QL/action/author_attestation","sign_citation":"https://pith.science/pith/7WRAHT5R7PQCGCSLZ4VZKPD6QL/action/citation_signature","submit_replication":"https://pith.science/pith/7WRAHT5R7PQCGCSLZ4VZKPD6QL/action/replication_record"}},"created_at":"2026-07-05T10:15:45.635395+00:00","updated_at":"2026-07-05T10:15:45.635395+00:00"}