{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZTFCOOCZZSESFCNDW2EESZXG3S","short_pith_number":"pith:ZTFCOOCZ","schema_version":"1.0","canonical_sha256":"ccca273859cc892289a3b6884966e6dca5bde063d752c3e733c277401c8a4f1a","source":{"kind":"arxiv","id":"2502.05782","version":1},"attestation_state":"computed","paper":{"title":"Quality Assurance for LLM-RAG Systems: Empirical Insights from Tourism Application Testing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Bestoun S. Ahmed, Firas Bayram, Ludwig Otto Baader, Peter Magnusson, Siri Jagstedt","submitted_at":"2025-02-09T05:53:03Z","abstract_excerpt":"This paper presents a comprehensive framework for testing and evaluating quality characteristics of Large Language Model (LLM) systems enhanced with Retrieval-Augmented Generation (RAG) in tourism applications. Through systematic empirical evaluation of three different LLM variants across multiple parameter configurations, we demonstrate the effectiveness of our testing methodology in assessing both functional correctness and extra-functional properties. Our framework implements 17 distinct metrics that encompass syntactic analysis, semantic evaluation, and behavioral evaluation through LLM ju"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.05782","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SE","submitted_at":"2025-02-09T05:53:03Z","cross_cats_sorted":[],"title_canon_sha256":"bf11d5d421d47d112b242c15c049e808bb0977b5e89dcbf29cbd66cdd18b7ced","abstract_canon_sha256":"b2f6a7cdc81c1b805e30342caed8f9e0bb167f874a8837e13bc47e5b4ece9b75"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:11:43.250899Z","signature_b64":"sDx/qO53Ibfh74mlPDHyk/LQ1gYsz2wd99f53XYPkU23aT/dCncDMUgp7r1NIOXhbw03S1lpIl3/hL0d4BOyBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ccca273859cc892289a3b6884966e6dca5bde063d752c3e733c277401c8a4f1a","last_reissued_at":"2026-07-05T10:11:43.250368Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:11:43.250368Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Quality Assurance for LLM-RAG Systems: Empirical Insights from Tourism Application Testing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Bestoun S. Ahmed, Firas Bayram, Ludwig Otto Baader, Peter Magnusson, Siri Jagstedt","submitted_at":"2025-02-09T05:53:03Z","abstract_excerpt":"This paper presents a comprehensive framework for testing and evaluating quality characteristics of Large Language Model (LLM) systems enhanced with Retrieval-Augmented Generation (RAG) in tourism applications. Through systematic empirical evaluation of three different LLM variants across multiple parameter configurations, we demonstrate the effectiveness of our testing methodology in assessing both functional correctness and extra-functional properties. Our framework implements 17 distinct metrics that encompass syntactic analysis, semantic evaluation, and behavioral evaluation through LLM ju"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.05782","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.05782/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.05782","created_at":"2026-07-05T10:11:43.250420+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.05782v1","created_at":"2026-07-05T10:11:43.250420+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.05782","created_at":"2026-07-05T10:11:43.250420+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZTFCOOCZZSES","created_at":"2026-07-05T10:11:43.250420+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZTFCOOCZZSESFCND","created_at":"2026-07-05T10:11:43.250420+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZTFCOOCZ","created_at":"2026-07-05T10:11:43.250420+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.08515","citing_title":"The Paradox of Stochasticity: Limited Creativity and Computational Decoupling in Temperature-Varied LLM Outputs of Structured Fictional Data","ref_index":4,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZTFCOOCZZSESFCNDW2EESZXG3S","json":"https://pith.science/pith/ZTFCOOCZZSESFCNDW2EESZXG3S.json","graph_json":"https://pith.science/api/pith-number/ZTFCOOCZZSESFCNDW2EESZXG3S/graph.json","events_json":"https://pith.science/api/pith-number/ZTFCOOCZZSESFCNDW2EESZXG3S/events.json","paper":"https://pith.science/paper/ZTFCOOCZ"},"agent_actions":{"view_html":"https://pith.science/pith/ZTFCOOCZZSESFCNDW2EESZXG3S","download_json":"https://pith.science/pith/ZTFCOOCZZSESFCNDW2EESZXG3S.json","view_paper":"https://pith.science/paper/ZTFCOOCZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.05782&json=true","fetch_graph":"https://pith.science/api/pith-number/ZTFCOOCZZSESFCNDW2EESZXG3S/graph.json","fetch_events":"https://pith.science/api/pith-number/ZTFCOOCZZSESFCNDW2EESZXG3S/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZTFCOOCZZSESFCNDW2EESZXG3S/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZTFCOOCZZSESFCNDW2EESZXG3S/action/storage_attestation","attest_author":"https://pith.science/pith/ZTFCOOCZZSESFCNDW2EESZXG3S/action/author_attestation","sign_citation":"https://pith.science/pith/ZTFCOOCZZSESFCNDW2EESZXG3S/action/citation_signature","submit_replication":"https://pith.science/pith/ZTFCOOCZZSESFCNDW2EESZXG3S/action/replication_record"}},"created_at":"2026-07-05T10:11:43.250420+00:00","updated_at":"2026-07-05T10:11:43.250420+00:00"}