{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:4RFRWGDTR72JMPWALQS2CM7MWS","short_pith_number":"pith:4RFRWGDT","schema_version":"1.0","canonical_sha256":"e44b1b18738ff4963ec05c25a133ecb489b819e43dda482cd46bb22af065e5bd","source":{"kind":"arxiv","id":"2309.12871","version":9},"attestation_state":"computed","paper":{"title":"AnglE-optimized Text Embeddings","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Jing Li, Xianming Li","submitted_at":"2023-09-22T13:52:42Z","abstract_excerpt":"High-quality text embedding is pivotal in improving semantic textual similarity (STS) tasks, which are crucial components in Large Language Model (LLM) applications. However, a common challenge existing text embedding models face is the problem of vanishing gradients, primarily due to their reliance on the cosine function in the optimization objective, which has saturation zones. To address this issue, this paper proposes a novel angle-optimized text embedding model called AnglE. The core idea of AnglE is to introduce angle optimization in a complex space. This novel approach effectively mitig"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.12871","kind":"arxiv","version":9},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-09-22T13:52:42Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"9546cf5158b9ff23c5880491e57eb7d5a7cd3793904a881f2344bee70b8713d6","abstract_canon_sha256":"211c454710f9090e5f2dff7ad5970d89bb0417dcdf3a20a42d78a0908710db9b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:55:31.656608Z","signature_b64":"GFWbuXHdw8nK9vLfNJZmLVbXiwUkn3ebhyZ3F9xaNDNVpQWSIm6dOwT3h/1bvhF/LZQgHYyBuc2/AAR8GLuzBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e44b1b18738ff4963ec05c25a133ecb489b819e43dda482cd46bb22af065e5bd","last_reissued_at":"2026-07-05T09:55:31.655860Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:55:31.655860Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AnglE-optimized Text Embeddings","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Jing Li, Xianming Li","submitted_at":"2023-09-22T13:52:42Z","abstract_excerpt":"High-quality text embedding is pivotal in improving semantic textual similarity (STS) tasks, which are crucial components in Large Language Model (LLM) applications. However, a common challenge existing text embedding models face is the problem of vanishing gradients, primarily due to their reliance on the cosine function in the optimization objective, which has saturation zones. To address this issue, this paper proposes a novel angle-optimized text embedding model called AnglE. The core idea of AnglE is to introduce angle optimization in a complex space. This novel approach effectively mitig"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.12871","kind":"arxiv","version":9},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.12871/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.12871","created_at":"2026-07-05T09:55:31.655941+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.12871v9","created_at":"2026-07-05T09:55:31.655941+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.12871","created_at":"2026-07-05T09:55:31.655941+00:00"},{"alias_kind":"pith_short_12","alias_value":"4RFRWGDTR72J","created_at":"2026-07-05T09:55:31.655941+00:00"},{"alias_kind":"pith_short_16","alias_value":"4RFRWGDTR72JMPWA","created_at":"2026-07-05T09:55:31.655941+00:00"},{"alias_kind":"pith_short_8","alias_value":"4RFRWGDT","created_at":"2026-07-05T09:55:31.655941+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21622","citing_title":"Evaluating Document-Tuned Transformer Representations for Person-level Mental Health Assessment","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07924","citing_title":"Decoupling Semantics and Logic: A Training-Free Coarse-to-Fine Pipeline for Video Retrieval-Augmented Generation","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05858","citing_title":"ReverseEOL: Improving Training-free Text Embeddings via Text Reversal in Decoder-only LLMs","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31142","citing_title":"On the Robustness of Multilingual Text Embedding Rankings Across Learning Tasks, Languages, and Benchmark Datasets","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02950","citing_title":"Kernel Affine Hull Machines as Compute-Efficient Encoders for Frozen Semantic Spaces","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2312.10997","citing_title":"Retrieval-Augmented Generation for Large Language Models: A Survey","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2409.02038","citing_title":"BEAVER: An Enterprise Benchmark for Text-to-SQL","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2411.12142","citing_title":"A Computational Method for Measuring \"Open Codes\" in Qualitative Analysis","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2412.14751","citing_title":"Query pipeline optimization for cancer patient question answering systems","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18653","citing_title":"Will It Go Viral? Grounding Micro-Video Popularity Prediction on the Open Web","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2603.28205","citing_title":"Beyond Cosine Similarity: Zero-Initialized Residual Complex Projection for Aspect-Based Sentiment Analysis","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2405.17428","citing_title":"NV-Embed: Improved Techniques for Training LLMs as Generalist Embedding Models","ref_index":130,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17771","citing_title":"SPENCE: A Syntactic Probe for Detecting Contamination in NL2SQL Benchmarks","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02950","citing_title":"Kernel Affine Hull Machines as Compute-Efficient Encoders for Frozen Semantic Spaces","ref_index":30,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4RFRWGDTR72JMPWALQS2CM7MWS","json":"https://pith.science/pith/4RFRWGDTR72JMPWALQS2CM7MWS.json","graph_json":"https://pith.science/api/pith-number/4RFRWGDTR72JMPWALQS2CM7MWS/graph.json","events_json":"https://pith.science/api/pith-number/4RFRWGDTR72JMPWALQS2CM7MWS/events.json","paper":"https://pith.science/paper/4RFRWGDT"},"agent_actions":{"view_html":"https://pith.science/pith/4RFRWGDTR72JMPWALQS2CM7MWS","download_json":"https://pith.science/pith/4RFRWGDTR72JMPWALQS2CM7MWS.json","view_paper":"https://pith.science/paper/4RFRWGDT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.12871&json=true","fetch_graph":"https://pith.science/api/pith-number/4RFRWGDTR72JMPWALQS2CM7MWS/graph.json","fetch_events":"https://pith.science/api/pith-number/4RFRWGDTR72JMPWALQS2CM7MWS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4RFRWGDTR72JMPWALQS2CM7MWS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4RFRWGDTR72JMPWALQS2CM7MWS/action/storage_attestation","attest_author":"https://pith.science/pith/4RFRWGDTR72JMPWALQS2CM7MWS/action/author_attestation","sign_citation":"https://pith.science/pith/4RFRWGDTR72JMPWALQS2CM7MWS/action/citation_signature","submit_replication":"https://pith.science/pith/4RFRWGDTR72JMPWALQS2CM7MWS/action/replication_record"}},"created_at":"2026-07-05T09:55:31.655941+00:00","updated_at":"2026-07-05T09:55:31.655941+00:00"}