{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LVF2JGUOTF2M3UQP6BD5PGA2KU","short_pith_number":"pith:LVF2JGUO","schema_version":"1.0","canonical_sha256":"5d4ba49a8e9974cdd20ff047d7981a552fd796f7e63d81eb0c461871281f123f","source":{"kind":"arxiv","id":"2403.00820","version":1},"attestation_state":"computed","paper":{"title":"Retrieval Augmented Generation Systems: Automatic Dataset Creation, Evaluation and Boolean Agent Setup","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.IR","authors_text":"Barbara Hammer, Philip Kenneweg, Tristan Kenneweg","submitted_at":"2024-02-26T12:56:17Z","abstract_excerpt":"Retrieval Augmented Generation (RAG) systems have seen huge popularity in augmenting Large-Language Model (LLM) outputs with domain specific and time sensitive data. Very recently a shift is happening from simple RAG setups that query a vector database for additional information with every user input to more sophisticated forms of RAG. However, different concrete approaches compete on mostly anecdotal evidence at the moment. In this paper we present a rigorous dataset creation and evaluation workflow to quantitatively compare different RAG strategies. We use a dataset created this way for the "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.00820","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.IR","submitted_at":"2024-02-26T12:56:17Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"44567cfa911647c171fda82f202154e959e0d644ecb074859349d573c98b1cc0","abstract_canon_sha256":"45ca63dd95fc8c7c3e6b760133c18c21c47057973d062d126f6ecb48efaf131f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:51:12.180223Z","signature_b64":"qTeTJL9hEfeYsrhmQ6MjUklgVsHpBcHO4kMxRe0ZAGxVIwpqt/qa7Q2TKygwI58fgOqxzdjDR8OSvSVN13xxDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5d4ba49a8e9974cdd20ff047d7981a552fd796f7e63d81eb0c461871281f123f","last_reissued_at":"2026-07-05T07:51:12.179748Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:51:12.179748Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Retrieval Augmented Generation Systems: Automatic Dataset Creation, Evaluation and Boolean Agent Setup","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.IR","authors_text":"Barbara Hammer, Philip Kenneweg, Tristan Kenneweg","submitted_at":"2024-02-26T12:56:17Z","abstract_excerpt":"Retrieval Augmented Generation (RAG) systems have seen huge popularity in augmenting Large-Language Model (LLM) outputs with domain specific and time sensitive data. Very recently a shift is happening from simple RAG setups that query a vector database for additional information with every user input to more sophisticated forms of RAG. However, different concrete approaches compete on mostly anecdotal evidence at the moment. In this paper we present a rigorous dataset creation and evaluation workflow to quantitatively compare different RAG strategies. We use a dataset created this way for the "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.00820","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.00820/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.00820","created_at":"2026-07-05T07:51:12.179804+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.00820v1","created_at":"2026-07-05T07:51:12.179804+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.00820","created_at":"2026-07-05T07:51:12.179804+00:00"},{"alias_kind":"pith_short_12","alias_value":"LVF2JGUOTF2M","created_at":"2026-07-05T07:51:12.179804+00:00"},{"alias_kind":"pith_short_16","alias_value":"LVF2JGUOTF2M3UQP","created_at":"2026-07-05T07:51:12.179804+00:00"},{"alias_kind":"pith_short_8","alias_value":"LVF2JGUO","created_at":"2026-07-05T07:51:12.179804+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2504.20119","citing_title":"Can LLMs Be Trusted for Evaluating RAG Systems? A Survey of Methods and Datasets","ref_index":24,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LVF2JGUOTF2M3UQP6BD5PGA2KU","json":"https://pith.science/pith/LVF2JGUOTF2M3UQP6BD5PGA2KU.json","graph_json":"https://pith.science/api/pith-number/LVF2JGUOTF2M3UQP6BD5PGA2KU/graph.json","events_json":"https://pith.science/api/pith-number/LVF2JGUOTF2M3UQP6BD5PGA2KU/events.json","paper":"https://pith.science/paper/LVF2JGUO"},"agent_actions":{"view_html":"https://pith.science/pith/LVF2JGUOTF2M3UQP6BD5PGA2KU","download_json":"https://pith.science/pith/LVF2JGUOTF2M3UQP6BD5PGA2KU.json","view_paper":"https://pith.science/paper/LVF2JGUO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.00820&json=true","fetch_graph":"https://pith.science/api/pith-number/LVF2JGUOTF2M3UQP6BD5PGA2KU/graph.json","fetch_events":"https://pith.science/api/pith-number/LVF2JGUOTF2M3UQP6BD5PGA2KU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LVF2JGUOTF2M3UQP6BD5PGA2KU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LVF2JGUOTF2M3UQP6BD5PGA2KU/action/storage_attestation","attest_author":"https://pith.science/pith/LVF2JGUOTF2M3UQP6BD5PGA2KU/action/author_attestation","sign_citation":"https://pith.science/pith/LVF2JGUOTF2M3UQP6BD5PGA2KU/action/citation_signature","submit_replication":"https://pith.science/pith/LVF2JGUOTF2M3UQP6BD5PGA2KU/action/replication_record"}},"created_at":"2026-07-05T07:51:12.179804+00:00","updated_at":"2026-07-05T07:51:12.179804+00:00"}