{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QSXCUGHZMGNBOEF6KNPOMOP367","short_pith_number":"pith:QSXCUGHZ","schema_version":"1.0","canonical_sha256":"84ae2a18f9619a1710be535ee639fbf7ff5c6b34a9a5bbd4258af982b60db386","source":{"kind":"arxiv","id":"2410.15944","version":1},"attestation_state":"computed","paper":{"title":"Developing Retrieval Augmented Generation (RAG) based LLM Systems from PDFs: An Experience Report","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.IR"],"primary_cat":"cs.SE","authors_text":"Ayman Asad Khan, Jussi Rasku, Kai Kristian Kemell, Md Toufique Hasan, Pekka Abrahamsson","submitted_at":"2024-10-21T12:21:49Z","abstract_excerpt":"This paper presents an experience report on the development of Retrieval Augmented Generation (RAG) systems using PDF documents as the primary data source. The RAG architecture combines generative capabilities of Large Language Models (LLMs) with the precision of information retrieval. This approach has the potential to redefine how we interact with and augment both structured and unstructured knowledge in generative models to enhance transparency, accuracy, and contextuality of responses. The paper details the end-to-end pipeline, from data collection, preprocessing, to retrieval indexing and"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.15944","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SE","submitted_at":"2024-10-21T12:21:49Z","cross_cats_sorted":["cs.AI","cs.IR"],"title_canon_sha256":"4a6198b81c88f060a79e76fb58fe780b6014e6bc867ba3811f2b50b8c2950cb7","abstract_canon_sha256":"fcbd2659f8d7d3d46746d729a59f490d5b6dfe0a498943342a6716f128275018"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:23:26.975493Z","signature_b64":"yW1pm6RtKwU4IfKDQh+qnAqllLrGMTe55k9IJpJzcaQoWMEOdfxJUSU2aNTgVAxRYp++bEmoljevDpbpErL5Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"84ae2a18f9619a1710be535ee639fbf7ff5c6b34a9a5bbd4258af982b60db386","last_reissued_at":"2026-07-05T09:23:26.975047Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:23:26.975047Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Developing Retrieval Augmented Generation (RAG) based LLM Systems from PDFs: An Experience Report","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.IR"],"primary_cat":"cs.SE","authors_text":"Ayman Asad Khan, Jussi Rasku, Kai Kristian Kemell, Md Toufique Hasan, Pekka Abrahamsson","submitted_at":"2024-10-21T12:21:49Z","abstract_excerpt":"This paper presents an experience report on the development of Retrieval Augmented Generation (RAG) systems using PDF documents as the primary data source. The RAG architecture combines generative capabilities of Large Language Models (LLMs) with the precision of information retrieval. This approach has the potential to redefine how we interact with and augment both structured and unstructured knowledge in generative models to enhance transparency, accuracy, and contextuality of responses. The paper details the end-to-end pipeline, from data collection, preprocessing, to retrieval indexing and"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.15944","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.15944/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.15944","created_at":"2026-07-05T09:23:26.975116+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.15944v1","created_at":"2026-07-05T09:23:26.975116+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.15944","created_at":"2026-07-05T09:23:26.975116+00:00"},{"alias_kind":"pith_short_12","alias_value":"QSXCUGHZMGNB","created_at":"2026-07-05T09:23:26.975116+00:00"},{"alias_kind":"pith_short_16","alias_value":"QSXCUGHZMGNBOEF6","created_at":"2026-07-05T09:23:26.975116+00:00"},{"alias_kind":"pith_short_8","alias_value":"QSXCUGHZ","created_at":"2026-07-05T09:23:26.975116+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.19779","citing_title":"ESGLens: An LLM-Based RAG Framework for Interactive ESG Report Analysis and Score Prediction","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QSXCUGHZMGNBOEF6KNPOMOP367","json":"https://pith.science/pith/QSXCUGHZMGNBOEF6KNPOMOP367.json","graph_json":"https://pith.science/api/pith-number/QSXCUGHZMGNBOEF6KNPOMOP367/graph.json","events_json":"https://pith.science/api/pith-number/QSXCUGHZMGNBOEF6KNPOMOP367/events.json","paper":"https://pith.science/paper/QSXCUGHZ"},"agent_actions":{"view_html":"https://pith.science/pith/QSXCUGHZMGNBOEF6KNPOMOP367","download_json":"https://pith.science/pith/QSXCUGHZMGNBOEF6KNPOMOP367.json","view_paper":"https://pith.science/paper/QSXCUGHZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.15944&json=true","fetch_graph":"https://pith.science/api/pith-number/QSXCUGHZMGNBOEF6KNPOMOP367/graph.json","fetch_events":"https://pith.science/api/pith-number/QSXCUGHZMGNBOEF6KNPOMOP367/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QSXCUGHZMGNBOEF6KNPOMOP367/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QSXCUGHZMGNBOEF6KNPOMOP367/action/storage_attestation","attest_author":"https://pith.science/pith/QSXCUGHZMGNBOEF6KNPOMOP367/action/author_attestation","sign_citation":"https://pith.science/pith/QSXCUGHZMGNBOEF6KNPOMOP367/action/citation_signature","submit_replication":"https://pith.science/pith/QSXCUGHZMGNBOEF6KNPOMOP367/action/replication_record"}},"created_at":"2026-07-05T09:23:26.975116+00:00","updated_at":"2026-07-05T09:23:26.975116+00:00"}