{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JLJ52YGN4JTI2JAAJ757ZAI2NY","short_pith_number":"pith:JLJ52YGN","schema_version":"1.0","canonical_sha256":"4ad3dd60cde2668d24004ffbfc811a6e38f0bbd06b57a10ffa9a415e4b135871","source":{"kind":"arxiv","id":"2410.13070","version":1},"attestation_state":"computed","paper":{"title":"Is Semantic Chunking Worth the Computational Cost?","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.IR"],"primary_cat":"cs.CL","authors_text":"Forrest Bao, Renyi Qu, Ruixuan Tu","submitted_at":"2024-10-16T21:53:48Z","abstract_excerpt":"Recent advances in Retrieval-Augmented Generation (RAG) systems have popularized semantic chunking, which aims to improve retrieval performance by dividing documents into semantically coherent segments. Despite its growing adoption, the actual benefits over simpler fixed-size chunking, where documents are split into consecutive, fixed-size segments, remain unclear. This study systematically evaluates the effectiveness of semantic chunking using three common retrieval-related tasks: document retrieval, evidence retrieval, and retrieval-based answer generation. The results show that the computat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.13070","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-16T21:53:48Z","cross_cats_sorted":["cs.IR"],"title_canon_sha256":"bc89933db02750e9b6f4a9f79aef1d43c05fc7ba198d5fd2d7d58915f83fbd55","abstract_canon_sha256":"19720bf5174767a54f519e4a3aa694f3598148d8c3b49386b85c62b8f46fb6a3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:21:54.866615Z","signature_b64":"jlVd7UKVad1n4NGhnlnpUtq4ZLd9bm0xVzzUlA6yNlf7bqNEVVw/M+oTm0ZogPkG85NZoHLWJg/ZTeen9uDkAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4ad3dd60cde2668d24004ffbfc811a6e38f0bbd06b57a10ffa9a415e4b135871","last_reissued_at":"2026-07-05T09:21:54.866131Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:21:54.866131Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Is Semantic Chunking Worth the Computational Cost?","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.IR"],"primary_cat":"cs.CL","authors_text":"Forrest Bao, Renyi Qu, Ruixuan Tu","submitted_at":"2024-10-16T21:53:48Z","abstract_excerpt":"Recent advances in Retrieval-Augmented Generation (RAG) systems have popularized semantic chunking, which aims to improve retrieval performance by dividing documents into semantically coherent segments. Despite its growing adoption, the actual benefits over simpler fixed-size chunking, where documents are split into consecutive, fixed-size segments, remain unclear. This study systematically evaluates the effectiveness of semantic chunking using three common retrieval-related tasks: document retrieval, evidence retrieval, and retrieval-based answer generation. The results show that the computat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.13070","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.13070/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.13070","created_at":"2026-07-05T09:21:54.866191+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.13070v1","created_at":"2026-07-05T09:21:54.866191+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.13070","created_at":"2026-07-05T09:21:54.866191+00:00"},{"alias_kind":"pith_short_12","alias_value":"JLJ52YGN4JTI","created_at":"2026-07-05T09:21:54.866191+00:00"},{"alias_kind":"pith_short_16","alias_value":"JLJ52YGN4JTI2JAA","created_at":"2026-07-05T09:21:54.866191+00:00"},{"alias_kind":"pith_short_8","alias_value":"JLJ52YGN","created_at":"2026-07-05T09:21:54.866191+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20900","citing_title":"Storyline Trees: Hierarchical Representations for Long-Form Narratives","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22203","citing_title":"Evaluation of Chunking Strategies for Effective Text Embedding in Low-Resource Language on Agricultural Documents","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JLJ52YGN4JTI2JAAJ757ZAI2NY","json":"https://pith.science/pith/JLJ52YGN4JTI2JAAJ757ZAI2NY.json","graph_json":"https://pith.science/api/pith-number/JLJ52YGN4JTI2JAAJ757ZAI2NY/graph.json","events_json":"https://pith.science/api/pith-number/JLJ52YGN4JTI2JAAJ757ZAI2NY/events.json","paper":"https://pith.science/paper/JLJ52YGN"},"agent_actions":{"view_html":"https://pith.science/pith/JLJ52YGN4JTI2JAAJ757ZAI2NY","download_json":"https://pith.science/pith/JLJ52YGN4JTI2JAAJ757ZAI2NY.json","view_paper":"https://pith.science/paper/JLJ52YGN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.13070&json=true","fetch_graph":"https://pith.science/api/pith-number/JLJ52YGN4JTI2JAAJ757ZAI2NY/graph.json","fetch_events":"https://pith.science/api/pith-number/JLJ52YGN4JTI2JAAJ757ZAI2NY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JLJ52YGN4JTI2JAAJ757ZAI2NY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JLJ52YGN4JTI2JAAJ757ZAI2NY/action/storage_attestation","attest_author":"https://pith.science/pith/JLJ52YGN4JTI2JAAJ757ZAI2NY/action/author_attestation","sign_citation":"https://pith.science/pith/JLJ52YGN4JTI2JAAJ757ZAI2NY/action/citation_signature","submit_replication":"https://pith.science/pith/JLJ52YGN4JTI2JAAJ757ZAI2NY/action/replication_record"}},"created_at":"2026-07-05T09:21:54.866191+00:00","updated_at":"2026-07-05T09:21:54.866191+00:00"}