{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LZGLMVOJ7IEHBBBYQLRGTRF3Q5","short_pith_number":"pith:LZGLMVOJ","schema_version":"1.0","canonical_sha256":"5e4cb655c9fa0870843882e269c4bb8761194096b2cd84d8effe94ed4dcf475f","source":{"kind":"arxiv","id":"2404.07221","version":2},"attestation_state":"computed","paper":{"title":"Improving Retrieval for RAG based Question Answering Models on Financial Documents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG","q-fin.GN"],"primary_cat":"cs.IR","authors_text":"Alyssa Lee, Eden Chung, Harsh Thakkar, Natan Vidra, Spurthi Setty","submitted_at":"2024-03-23T00:49:40Z","abstract_excerpt":"The effectiveness of Large Language Models (LLMs) in generating accurate responses relies heavily on the quality of input provided, particularly when employing Retrieval Augmented Generation (RAG) techniques. RAG enhances LLMs by sourcing the most relevant text chunk(s) to base queries upon. Despite the significant advancements in LLMs' response quality in recent years, users may still encounter inaccuracies or irrelevant answers; these issues often stem from suboptimal text chunk retrieval by RAG rather than the inherent capabilities of LLMs. To augment the efficacy of LLMs, it is crucial to "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.07221","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.IR","submitted_at":"2024-03-23T00:49:40Z","cross_cats_sorted":["cs.CL","cs.LG","q-fin.GN"],"title_canon_sha256":"652021da7ae585fdfc33c5138be0dc42446f840fa832e544b5f650f7137c16d4","abstract_canon_sha256":"c9e0673ff52047d3ea1e2347a67617c5efde97933c04fd5ae1ee895942eb037c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:50:55.568620Z","signature_b64":"mOYEUHWgoGXDFGM3Ga7lPavtkFH98VjhXO25qlxGztCrQuCEu4p5eZKio6F/ybN8a7zDrhIAeV+Jv3vpje3mAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5e4cb655c9fa0870843882e269c4bb8761194096b2cd84d8effe94ed4dcf475f","last_reissued_at":"2026-07-05T08:50:55.568081Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:50:55.568081Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Improving Retrieval for RAG based Question Answering Models on Financial Documents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG","q-fin.GN"],"primary_cat":"cs.IR","authors_text":"Alyssa Lee, Eden Chung, Harsh Thakkar, Natan Vidra, Spurthi Setty","submitted_at":"2024-03-23T00:49:40Z","abstract_excerpt":"The effectiveness of Large Language Models (LLMs) in generating accurate responses relies heavily on the quality of input provided, particularly when employing Retrieval Augmented Generation (RAG) techniques. RAG enhances LLMs by sourcing the most relevant text chunk(s) to base queries upon. Despite the significant advancements in LLMs' response quality in recent years, users may still encounter inaccuracies or irrelevant answers; these issues often stem from suboptimal text chunk retrieval by RAG rather than the inherent capabilities of LLMs. To augment the efficacy of LLMs, it is crucial to "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.07221","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.07221/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.07221","created_at":"2026-07-05T08:50:55.568137+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.07221v2","created_at":"2026-07-05T08:50:55.568137+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.07221","created_at":"2026-07-05T08:50:55.568137+00:00"},{"alias_kind":"pith_short_12","alias_value":"LZGLMVOJ7IEH","created_at":"2026-07-05T08:50:55.568137+00:00"},{"alias_kind":"pith_short_16","alias_value":"LZGLMVOJ7IEHBBBY","created_at":"2026-07-05T08:50:55.568137+00:00"},{"alias_kind":"pith_short_8","alias_value":"LZGLMVOJ","created_at":"2026-07-05T08:50:55.568137+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.25030","citing_title":"MimirRAG: A Multi-Agent RAG Framework for Financial Data Retrieval with Metadata Integration","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00869","citing_title":"Enhancing LLM Metacognition via Cognitive Pairwise Training","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05818","citing_title":"LeakDojo: Decoding the Leakage Threats of RAG Systems","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05459","citing_title":"Privacy Without Losing Place: A Paradigm for Private Retrieval in Spatial RAGs","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14222","citing_title":"Adaptive Query Routing: A Tier-Based Framework for Hybrid Retrieval Across Financial, Legal, and Medical Documents","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LZGLMVOJ7IEHBBBYQLRGTRF3Q5","json":"https://pith.science/pith/LZGLMVOJ7IEHBBBYQLRGTRF3Q5.json","graph_json":"https://pith.science/api/pith-number/LZGLMVOJ7IEHBBBYQLRGTRF3Q5/graph.json","events_json":"https://pith.science/api/pith-number/LZGLMVOJ7IEHBBBYQLRGTRF3Q5/events.json","paper":"https://pith.science/paper/LZGLMVOJ"},"agent_actions":{"view_html":"https://pith.science/pith/LZGLMVOJ7IEHBBBYQLRGTRF3Q5","download_json":"https://pith.science/pith/LZGLMVOJ7IEHBBBYQLRGTRF3Q5.json","view_paper":"https://pith.science/paper/LZGLMVOJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.07221&json=true","fetch_graph":"https://pith.science/api/pith-number/LZGLMVOJ7IEHBBBYQLRGTRF3Q5/graph.json","fetch_events":"https://pith.science/api/pith-number/LZGLMVOJ7IEHBBBYQLRGTRF3Q5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LZGLMVOJ7IEHBBBYQLRGTRF3Q5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LZGLMVOJ7IEHBBBYQLRGTRF3Q5/action/storage_attestation","attest_author":"https://pith.science/pith/LZGLMVOJ7IEHBBBYQLRGTRF3Q5/action/author_attestation","sign_citation":"https://pith.science/pith/LZGLMVOJ7IEHBBBYQLRGTRF3Q5/action/citation_signature","submit_replication":"https://pith.science/pith/LZGLMVOJ7IEHBBBYQLRGTRF3Q5/action/replication_record"}},"created_at":"2026-07-05T08:50:55.568137+00:00","updated_at":"2026-07-05T08:50:55.568137+00:00"}