{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KJV5M3RC3NKF7CUJ33KEPKE3PY","short_pith_number":"pith:KJV5M3RC","schema_version":"1.0","canonical_sha256":"526bd66e22db545f8a89ded447a89b7e1c8e7b7ad20c76ce9adcd94e3ba430a2","source":{"kind":"arxiv","id":"2409.03759","version":1},"attestation_state":"computed","paper":{"title":"VERA: Validation and Evaluation of Retrieval-Augmented Systems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.IR","authors_text":"Adi Banerjee, Juan Pablo De la Cruz Weinstein, Laurent Mombaerts, Tarik Borogovac, Tianyu Ding, Yunhong Li","submitted_at":"2024-08-16T21:59:59Z","abstract_excerpt":"The increasing use of Retrieval-Augmented Generation (RAG) systems in various applications necessitates stringent protocols to ensure RAG systems accuracy, safety, and alignment with user intentions. In this paper, we introduce VERA (Validation and Evaluation of Retrieval-Augmented Systems), a framework designed to enhance the transparency and reliability of outputs from large language models (LLMs) that utilize retrieved information. VERA improves the way we evaluate RAG systems in two important ways: (1) it introduces a cross-encoder based mechanism that encompasses a set of multidimensional"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.03759","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.IR","submitted_at":"2024-08-16T21:59:59Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"adbb3741c5ef1b43d0a2f35f40b7ff7d900cc0b86c22e322e9fbbc5305fca41e","abstract_canon_sha256":"b096a6755ecfd340e22d629b3faa45ad5cee6eda04e24b06e0c51c21d7f1808e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:03:56.598122Z","signature_b64":"I43Vrkqu0r0lyYdVcqomXhJh8cjCS7IZAMnDjTPVxi2ygEAzayIGCEc0f5m8aIgaZHz5kHedBa5CyOBOsjvoBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"526bd66e22db545f8a89ded447a89b7e1c8e7b7ad20c76ce9adcd94e3ba430a2","last_reissued_at":"2026-07-05T09:03:56.597623Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:03:56.597623Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VERA: Validation and Evaluation of Retrieval-Augmented Systems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.IR","authors_text":"Adi Banerjee, Juan Pablo De la Cruz Weinstein, Laurent Mombaerts, Tarik Borogovac, Tianyu Ding, Yunhong Li","submitted_at":"2024-08-16T21:59:59Z","abstract_excerpt":"The increasing use of Retrieval-Augmented Generation (RAG) systems in various applications necessitates stringent protocols to ensure RAG systems accuracy, safety, and alignment with user intentions. In this paper, we introduce VERA (Validation and Evaluation of Retrieval-Augmented Systems), a framework designed to enhance the transparency and reliability of outputs from large language models (LLMs) that utilize retrieved information. VERA improves the way we evaluate RAG systems in two important ways: (1) it introduces a cross-encoder based mechanism that encompasses a set of multidimensional"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.03759","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.03759/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.03759","created_at":"2026-07-05T09:03:56.597683+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.03759v1","created_at":"2026-07-05T09:03:56.597683+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.03759","created_at":"2026-07-05T09:03:56.597683+00:00"},{"alias_kind":"pith_short_12","alias_value":"KJV5M3RC3NKF","created_at":"2026-07-05T09:03:56.597683+00:00"},{"alias_kind":"pith_short_16","alias_value":"KJV5M3RC3NKF7CUJ","created_at":"2026-07-05T09:03:56.597683+00:00"},{"alias_kind":"pith_short_8","alias_value":"KJV5M3RC","created_at":"2026-07-05T09:03:56.597683+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.17943","citing_title":"A Benchmark Construction and Evaluation Framework for Specialist Domains: Case Study on Defense-related Documents","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18234","citing_title":"Evaluating Multi-Hop Reasoning in RAG Systems: A Comparison of LLM-Based Retriever Evaluation Strategies","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KJV5M3RC3NKF7CUJ33KEPKE3PY","json":"https://pith.science/pith/KJV5M3RC3NKF7CUJ33KEPKE3PY.json","graph_json":"https://pith.science/api/pith-number/KJV5M3RC3NKF7CUJ33KEPKE3PY/graph.json","events_json":"https://pith.science/api/pith-number/KJV5M3RC3NKF7CUJ33KEPKE3PY/events.json","paper":"https://pith.science/paper/KJV5M3RC"},"agent_actions":{"view_html":"https://pith.science/pith/KJV5M3RC3NKF7CUJ33KEPKE3PY","download_json":"https://pith.science/pith/KJV5M3RC3NKF7CUJ33KEPKE3PY.json","view_paper":"https://pith.science/paper/KJV5M3RC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.03759&json=true","fetch_graph":"https://pith.science/api/pith-number/KJV5M3RC3NKF7CUJ33KEPKE3PY/graph.json","fetch_events":"https://pith.science/api/pith-number/KJV5M3RC3NKF7CUJ33KEPKE3PY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KJV5M3RC3NKF7CUJ33KEPKE3PY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KJV5M3RC3NKF7CUJ33KEPKE3PY/action/storage_attestation","attest_author":"https://pith.science/pith/KJV5M3RC3NKF7CUJ33KEPKE3PY/action/author_attestation","sign_citation":"https://pith.science/pith/KJV5M3RC3NKF7CUJ33KEPKE3PY/action/citation_signature","submit_replication":"https://pith.science/pith/KJV5M3RC3NKF7CUJ33KEPKE3PY/action/replication_record"}},"created_at":"2026-07-05T09:03:56.597683+00:00","updated_at":"2026-07-05T09:03:56.597683+00:00"}