{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:WJ22NHO5JPFD6HR4H4D3LOG6GB","short_pith_number":"pith:WJ22NHO5","schema_version":"1.0","canonical_sha256":"b275a69ddd4bca3f1e3c3f07b5b8de304244286ba5ccb274c0a6837f796cf9fb","source":{"kind":"arxiv","id":"2310.19923","version":4},"attestation_state":"computed","paper":{"title":"Jina Embeddings 2: 8192-Token General-Purpose Text Embeddings for Long Documents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Alaeddine Abdessalem, Bo Wang, Georgios Mastrapas, Han Xiao, Isabelle Mohr, Jackmin Ong, Maximilian Werk, Michael G\\\"unther, Mohammad Kalim Akram, Nan Wang, Saba Sturua, Susana Guzman, Tanguy Abel","submitted_at":"2023-10-30T18:35:30Z","abstract_excerpt":"Text embedding models have emerged as powerful tools for transforming sentences into fixed-sized feature vectors that encapsulate semantic information. While these models are essential for tasks like information retrieval, semantic clustering, and text re-ranking, most existing open-source models, especially those built on architectures like BERT, struggle to represent lengthy documents and often resort to truncation. One common approach to mitigate this challenge involves splitting documents into smaller paragraphs for embedding. However, this strategy results in a much larger set of vectors,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.19923","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-10-30T18:35:30Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"1ac5b542e24276ddf52a3f1167c7f9e5eeaa87ccc1e2317038a892409a08837c","abstract_canon_sha256":"d2a085badc08cf53f1b8eaa221b1bead3c0249c199cebd526039e256cd33cd5c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:40:58.435241Z","signature_b64":"viRTbksyhFICmTTNH/cBF4rVf62bHo8RevVeYNO4iMOnW3D0aehLsx2PGeBfkCeNhfBoplwEw3t8gAgIs5RBDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b275a69ddd4bca3f1e3c3f07b5b8de304244286ba5ccb274c0a6837f796cf9fb","last_reissued_at":"2026-07-05T07:40:58.434733Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:40:58.434733Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Jina Embeddings 2: 8192-Token General-Purpose Text Embeddings for Long Documents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Alaeddine Abdessalem, Bo Wang, Georgios Mastrapas, Han Xiao, Isabelle Mohr, Jackmin Ong, Maximilian Werk, Michael G\\\"unther, Mohammad Kalim Akram, Nan Wang, Saba Sturua, Susana Guzman, Tanguy Abel","submitted_at":"2023-10-30T18:35:30Z","abstract_excerpt":"Text embedding models have emerged as powerful tools for transforming sentences into fixed-sized feature vectors that encapsulate semantic information. While these models are essential for tasks like information retrieval, semantic clustering, and text re-ranking, most existing open-source models, especially those built on architectures like BERT, struggle to represent lengthy documents and often resort to truncation. One common approach to mitigate this challenge involves splitting documents into smaller paragraphs for embedding. However, this strategy results in a much larger set of vectors,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.19923","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.19923/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.19923","created_at":"2026-07-05T07:40:58.434790+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.19923v4","created_at":"2026-07-05T07:40:58.434790+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.19923","created_at":"2026-07-05T07:40:58.434790+00:00"},{"alias_kind":"pith_short_12","alias_value":"WJ22NHO5JPFD","created_at":"2026-07-05T07:40:58.434790+00:00"},{"alias_kind":"pith_short_16","alias_value":"WJ22NHO5JPFD6HR4","created_at":"2026-07-05T07:40:58.434790+00:00"},{"alias_kind":"pith_short_8","alias_value":"WJ22NHO5","created_at":"2026-07-05T07:40:58.434790+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23642","citing_title":"Improving Long-Context Retrieval with Multi-Prefix Embedding","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03027","citing_title":"SEA-Embedding: Open and Reproducible Text Embeddings for Southeast Asia","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01561","citing_title":"S-SPPO: Semantic-Calibrated Self-Play Preference Optimization","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24297","citing_title":"Benchmarking Patent Embeddings: A Multi-Task Evaluation of 22 Models Across Retrieval, Classification, and Clustering","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28742","citing_title":"CORE: Contrastive Reflection Enables Rapid Improvements in Reasoning","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2509.21743","citing_title":"Retrieval-of-Thought: Efficient Reasoning via Reusing Thoughts","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2401.15391","citing_title":"MultiHop-RAG: Benchmarking Retrieval-Augmented Generation for Multi-Hop Queries","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14503","citing_title":"Not All RAGs Are Created Equal: A Component-Wise Empirical Study for Software Engineering Tasks","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06455","citing_title":"PrefixGuard: From LLM-Agent Traces to Online Failure-Warning Monitors","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05251","citing_title":"Identifier-Free Code Embedding Models for Scalable Search","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WJ22NHO5JPFD6HR4H4D3LOG6GB","json":"https://pith.science/pith/WJ22NHO5JPFD6HR4H4D3LOG6GB.json","graph_json":"https://pith.science/api/pith-number/WJ22NHO5JPFD6HR4H4D3LOG6GB/graph.json","events_json":"https://pith.science/api/pith-number/WJ22NHO5JPFD6HR4H4D3LOG6GB/events.json","paper":"https://pith.science/paper/WJ22NHO5"},"agent_actions":{"view_html":"https://pith.science/pith/WJ22NHO5JPFD6HR4H4D3LOG6GB","download_json":"https://pith.science/pith/WJ22NHO5JPFD6HR4H4D3LOG6GB.json","view_paper":"https://pith.science/paper/WJ22NHO5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.19923&json=true","fetch_graph":"https://pith.science/api/pith-number/WJ22NHO5JPFD6HR4H4D3LOG6GB/graph.json","fetch_events":"https://pith.science/api/pith-number/WJ22NHO5JPFD6HR4H4D3LOG6GB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WJ22NHO5JPFD6HR4H4D3LOG6GB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WJ22NHO5JPFD6HR4H4D3LOG6GB/action/storage_attestation","attest_author":"https://pith.science/pith/WJ22NHO5JPFD6HR4H4D3LOG6GB/action/author_attestation","sign_citation":"https://pith.science/pith/WJ22NHO5JPFD6HR4H4D3LOG6GB/action/citation_signature","submit_replication":"https://pith.science/pith/WJ22NHO5JPFD6HR4H4D3LOG6GB/action/replication_record"}},"created_at":"2026-07-05T07:40:58.434790+00:00","updated_at":"2026-07-05T07:40:58.434790+00:00"}