{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TVCSZTMLTZSHZWR3BBH5LN5SPX","short_pith_number":"pith:TVCSZTML","schema_version":"1.0","canonical_sha256":"9d452ccd8b9e647cda3b084fd5b7b27dd939ad9d0d67d691bccfa2f9f598811d","source":{"kind":"arxiv","id":"2402.05131","version":3},"attestation_state":"computed","paper":{"title":"Financial Report Chunking for Effective Retrieval Augmented Generation","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Antonio Jimeno Yepes, Jan Milczek, Renyu Li, Sebastian Laverde, Yao You","submitted_at":"2024-02-05T22:35:42Z","abstract_excerpt":"Chunking information is a key step in Retrieval Augmented Generation (RAG). Current research primarily centers on paragraph-level chunking. This approach treats all texts as equal and neglects the information contained in the structure of documents. We propose an expanded approach to chunk documents by moving beyond mere paragraph-level chunking to chunk primary by structural element components of documents. Dissecting documents into these constituent elements creates a new way to chunk documents that yields the best chunk size without tuning. We introduce a novel framework that evaluates how "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.05131","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2024-02-05T22:35:42Z","cross_cats_sorted":[],"title_canon_sha256":"3b96aaf695c258168288fe49cca6d30cad24aaff1269e804c189ceba07de34a8","abstract_canon_sha256":"8d07651e90654929ef90ce8acf9d17aed541609d53e292bfe4f9c34682f786ee"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:56:59.282555Z","signature_b64":"6ij/FcSPtFRYG+nvg1usued+YaiGlyEoSPPqbQcnvmCpba5P18uM9WzRmv2aKbCIOW54dZKTDBXrC9nRNyVjBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9d452ccd8b9e647cda3b084fd5b7b27dd939ad9d0d67d691bccfa2f9f598811d","last_reissued_at":"2026-07-05T07:56:59.281952Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:56:59.281952Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Financial Report Chunking for Effective Retrieval Augmented Generation","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Antonio Jimeno Yepes, Jan Milczek, Renyu Li, Sebastian Laverde, Yao You","submitted_at":"2024-02-05T22:35:42Z","abstract_excerpt":"Chunking information is a key step in Retrieval Augmented Generation (RAG). Current research primarily centers on paragraph-level chunking. This approach treats all texts as equal and neglects the information contained in the structure of documents. We propose an expanded approach to chunk documents by moving beyond mere paragraph-level chunking to chunk primary by structural element components of documents. Dissecting documents into these constituent elements creates a new way to chunk documents that yields the best chunk size without tuning. We introduce a novel framework that evaluates how "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.05131","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.05131/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.05131","created_at":"2026-07-05T07:56:59.282030+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.05131v3","created_at":"2026-07-05T07:56:59.282030+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.05131","created_at":"2026-07-05T07:56:59.282030+00:00"},{"alias_kind":"pith_short_12","alias_value":"TVCSZTMLTZSH","created_at":"2026-07-05T07:56:59.282030+00:00"},{"alias_kind":"pith_short_16","alias_value":"TVCSZTMLTZSHZWR3","created_at":"2026-07-05T07:56:59.282030+00:00"},{"alias_kind":"pith_short_8","alias_value":"TVCSZTML","created_at":"2026-07-05T07:56:59.282030+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19887","citing_title":"FinRED: An Expert-Guided Benchmark Generation and Evaluation Framework for Financial LLM Red-Teaming","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13100","citing_title":"LEDGER: A Long-Context Benchmark of Corporate Annual Reports for Grounded Financial Retrieval and Extraction","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01852","citing_title":"Evaluating Chunking Strategies for Retrieval-Augmented Generation on Academic Texts","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12916","citing_title":"SHM-Agents: A Generalist-Specialist Integrated Agent System for Structural Health Monitoring","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25030","citing_title":"MimirRAG: A Multi-Agent RAG Framework for Financial Data Retrieval with Metadata Integration","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2506.20821","citing_title":"MultiFinRAG: An Optimized Multimodal Retrieval-Augmented Generation (RAG) Framework for Financial Question Answering","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2509.09544","citing_title":"MetaGraph: A Large-Scale Meta-Analysis of GenAI in Financial NLP (2022-2025)","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17979","citing_title":"Architecture Matters More Than Scale: A Comparative Study of Retrieval and Memory Augmentation for Financial QA Under SME Compute Constraints","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22861","citing_title":"IntrAgent: An LLM Agent for Content-Grounded Information Retrieval through Literature Review","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17979","citing_title":"Architecture Matters More Than Scale: A Comparative Study of Retrieval and Memory Augmentation for Financial QA Under SME Compute Constraints","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12047","citing_title":"Empirical Evaluation of PDF Parsing and Chunking for Financial Question Answering with RAG","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07403","citing_title":"RefineRAG: Word-Level Poisoning Attacks via Retriever-Guided Text Refinement","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14222","citing_title":"Adaptive Query Routing: A Tier-Based Framework for Hybrid Retrieval Across Financial, Legal, and Medical Documents","ref_index":52,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TVCSZTMLTZSHZWR3BBH5LN5SPX","json":"https://pith.science/pith/TVCSZTMLTZSHZWR3BBH5LN5SPX.json","graph_json":"https://pith.science/api/pith-number/TVCSZTMLTZSHZWR3BBH5LN5SPX/graph.json","events_json":"https://pith.science/api/pith-number/TVCSZTMLTZSHZWR3BBH5LN5SPX/events.json","paper":"https://pith.science/paper/TVCSZTML"},"agent_actions":{"view_html":"https://pith.science/pith/TVCSZTMLTZSHZWR3BBH5LN5SPX","download_json":"https://pith.science/pith/TVCSZTMLTZSHZWR3BBH5LN5SPX.json","view_paper":"https://pith.science/paper/TVCSZTML","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.05131&json=true","fetch_graph":"https://pith.science/api/pith-number/TVCSZTMLTZSHZWR3BBH5LN5SPX/graph.json","fetch_events":"https://pith.science/api/pith-number/TVCSZTMLTZSHZWR3BBH5LN5SPX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TVCSZTMLTZSHZWR3BBH5LN5SPX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TVCSZTMLTZSHZWR3BBH5LN5SPX/action/storage_attestation","attest_author":"https://pith.science/pith/TVCSZTMLTZSHZWR3BBH5LN5SPX/action/author_attestation","sign_citation":"https://pith.science/pith/TVCSZTMLTZSHZWR3BBH5LN5SPX/action/citation_signature","submit_replication":"https://pith.science/pith/TVCSZTMLTZSHZWR3BBH5LN5SPX/action/replication_record"}},"created_at":"2026-07-05T07:56:59.282030+00:00","updated_at":"2026-07-05T07:56:59.282030+00:00"}