{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:K44LOKHUGH645F3673YX5V5ZXO","short_pith_number":"pith:K44LOKHU","schema_version":"1.0","canonical_sha256":"5738b728f431fdce977efef17ed7b9bbb71a8f3ec5c1c665755661f5f6df6696","source":{"kind":"arxiv","id":"2409.04701","version":3},"attestation_state":"computed","paper":{"title":"Late Chunking: Contextual Chunk Embeddings Using Long-Context Embedding Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.IR"],"primary_cat":"cs.CL","authors_text":"Bo Wang, Daniel James Williams, Han Xiao, Isabelle Mohr, Michael G\\\"unther","submitted_at":"2024-09-07T03:54:46Z","abstract_excerpt":"Many use cases require retrieving smaller portions of text, and dense vector-based retrieval systems often perform better with shorter text segments, as the semantics are less likely to be over-compressed in the embeddings. Consequently, practitioners often split text documents into smaller chunks and encode them separately. However, chunk embeddings created in this way can lose contextual information from surrounding chunks, resulting in sub-optimal representations. In this paper, we introduce a novel method called late chunking, which leverages long context embedding models to first embed al"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.04701","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-09-07T03:54:46Z","cross_cats_sorted":["cs.IR"],"title_canon_sha256":"44f06378e1e7b6b2f5b0a1716052117c8d8dc20449316e05a2c4a70c56a1eb83","abstract_canon_sha256":"d25593c03c79ea413870386dea7c501729c64c1acc5bb47c79d385a536fe0d2f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:33:02.260828Z","signature_b64":"9IvDJgrml+7qF3LRwsp2mJVUXBNWl5hMq4VTvR7SDTz8qh9kdCoWKnB8yAFfcT13QD5eiZkmGg93DADGwS/vDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5738b728f431fdce977efef17ed7b9bbb71a8f3ec5c1c665755661f5f6df6696","last_reissued_at":"2026-07-05T11:33:02.260178Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:33:02.260178Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Late Chunking: Contextual Chunk Embeddings Using Long-Context Embedding Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.IR"],"primary_cat":"cs.CL","authors_text":"Bo Wang, Daniel James Williams, Han Xiao, Isabelle Mohr, Michael G\\\"unther","submitted_at":"2024-09-07T03:54:46Z","abstract_excerpt":"Many use cases require retrieving smaller portions of text, and dense vector-based retrieval systems often perform better with shorter text segments, as the semantics are less likely to be over-compressed in the embeddings. Consequently, practitioners often split text documents into smaller chunks and encode them separately. However, chunk embeddings created in this way can lose contextual information from surrounding chunks, resulting in sub-optimal representations. In this paper, we introduce a novel method called late chunking, which leverages long context embedding models to first embed al"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.04701","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.04701/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.04701","created_at":"2026-07-05T11:33:02.260267+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.04701v3","created_at":"2026-07-05T11:33:02.260267+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.04701","created_at":"2026-07-05T11:33:02.260267+00:00"},{"alias_kind":"pith_short_12","alias_value":"K44LOKHUGH64","created_at":"2026-07-05T11:33:02.260267+00:00"},{"alias_kind":"pith_short_16","alias_value":"K44LOKHUGH645F36","created_at":"2026-07-05T11:33:02.260267+00:00"},{"alias_kind":"pith_short_8","alias_value":"K44LOKHU","created_at":"2026-07-05T11:33:02.260267+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05927","citing_title":"CMDR: Contextual Multimodal Document Retrieval","ref_index":18,"is_internal_anchor":true},{"citing_arxiv_id":"2606.23642","citing_title":"Improving Long-Context Retrieval with Multi-Prefix Embedding","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18781","citing_title":"Lost in a Single Vector: Improving Long-Document Retrieval with Chunk Evidence Aggregation","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01852","citing_title":"Evaluating Chunking Strategies for Retrieval-Augmented Generation on Academic Texts","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01852","citing_title":"Evaluating Chunking Strategies for Retrieval-Augmented Generation on Academic Texts","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06906","citing_title":"EASE-TTT: Evidence-Aligned Selective Test-Time Training for Long-Context Question Answering","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01240","citing_title":"Efficient RAG with Intent-Aware Retrieval and Semantics-Preserving Chunking","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00881","citing_title":"Chunking Methods on Retrieval-Augmented Generation - Effectiveness Evaluation Against Computational Cost and Limitations","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22247","citing_title":"IdioLink: Retrieving Meaning Beyond Words Across Idiomatic and Literal Expressions","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2507.00994","citing_title":"Should We Still Pretrain Encoders with Masked Language Modeling?","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20849","citing_title":"SPIRE: Structure-Preserving Interpretable Retrieval of Evidence","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10296","citing_title":"Qwen Goes Brrr: Off-the-Shelf RAG for Ukrainian Multi-Domain Document Understanding","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24334","citing_title":"Reducing Redundancy in Retrieval-Augmented Generation through Chunk Filtering","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10167","citing_title":"Visual Late Chunking: An Empirical Study of Contextual Chunking for Efficient Visual Document Retrieval","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K44LOKHUGH645F3673YX5V5ZXO","json":"https://pith.science/pith/K44LOKHUGH645F3673YX5V5ZXO.json","graph_json":"https://pith.science/api/pith-number/K44LOKHUGH645F3673YX5V5ZXO/graph.json","events_json":"https://pith.science/api/pith-number/K44LOKHUGH645F3673YX5V5ZXO/events.json","paper":"https://pith.science/paper/K44LOKHU"},"agent_actions":{"view_html":"https://pith.science/pith/K44LOKHUGH645F3673YX5V5ZXO","download_json":"https://pith.science/pith/K44LOKHUGH645F3673YX5V5ZXO.json","view_paper":"https://pith.science/paper/K44LOKHU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.04701&json=true","fetch_graph":"https://pith.science/api/pith-number/K44LOKHUGH645F3673YX5V5ZXO/graph.json","fetch_events":"https://pith.science/api/pith-number/K44LOKHUGH645F3673YX5V5ZXO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K44LOKHUGH645F3673YX5V5ZXO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K44LOKHUGH645F3673YX5V5ZXO/action/storage_attestation","attest_author":"https://pith.science/pith/K44LOKHUGH645F3673YX5V5ZXO/action/author_attestation","sign_citation":"https://pith.science/pith/K44LOKHUGH645F3673YX5V5ZXO/action/citation_signature","submit_replication":"https://pith.science/pith/K44LOKHUGH645F3673YX5V5ZXO/action/replication_record"}},"created_at":"2026-07-05T11:33:02.260267+00:00","updated_at":"2026-07-05T11:33:02.260267+00:00"}