{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:UE3QTNKNC2LDAF6LMEDWSGIKEG","short_pith_number":"pith:UE3QTNKN","schema_version":"1.0","canonical_sha256":"a13709b54d16963017cb610769190a21af9c61a91225f903c1a1a5accc374597","source":{"kind":"arxiv","id":"2506.20128","version":1},"attestation_state":"computed","paper":{"title":"CCRS: A Zero-Shot LLM-as-a-Judge Framework for Comprehensive RAG Evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Aashiq Muhamed","submitted_at":"2025-06-25T04:49:03Z","abstract_excerpt":"RAG systems enhance LLMs by incorporating external knowledge, which is crucial for domains that demand factual accuracy and up-to-date information. However, evaluating the multifaceted quality of RAG outputs, spanning aspects such as contextual coherence, query relevance, factual correctness, and informational completeness, poses significant challenges. Existing evaluation methods often rely on simple lexical overlap metrics, which are inadequate for capturing these nuances, or involve complex multi-stage pipelines with intermediate steps like claim extraction or require finetuning specialized"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.20128","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-25T04:49:03Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"58bf829d4f2884ad6a842cd9d0b3ee9a3acc3ced23c69e9b2dc03558dffedb56","abstract_canon_sha256":"a49ca7a6e4285ef3c9119ac65669d813fe58816b00226e5a5daf5194edd6b379"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:27:00.774075Z","signature_b64":"JkdVir05gSlRXIEsZo19Ywht5JBGb7EnugW2xY2NleMLVAm7x47BFUhWq/SrI5rphbKvUD3WAT7as1GgzTBRCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a13709b54d16963017cb610769190a21af9c61a91225f903c1a1a5accc374597","last_reissued_at":"2026-07-05T11:27:00.773592Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:27:00.773592Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CCRS: A Zero-Shot LLM-as-a-Judge Framework for Comprehensive RAG Evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Aashiq Muhamed","submitted_at":"2025-06-25T04:49:03Z","abstract_excerpt":"RAG systems enhance LLMs by incorporating external knowledge, which is crucial for domains that demand factual accuracy and up-to-date information. However, evaluating the multifaceted quality of RAG outputs, spanning aspects such as contextual coherence, query relevance, factual correctness, and informational completeness, poses significant challenges. Existing evaluation methods often rely on simple lexical overlap metrics, which are inadequate for capturing these nuances, or involve complex multi-stage pipelines with intermediate steps like claim extraction or require finetuning specialized"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.20128","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.20128/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.20128","created_at":"2026-07-05T11:27:00.773651+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.20128v1","created_at":"2026-07-05T11:27:00.773651+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.20128","created_at":"2026-07-05T11:27:00.773651+00:00"},{"alias_kind":"pith_short_12","alias_value":"UE3QTNKNC2LD","created_at":"2026-07-05T11:27:00.773651+00:00"},{"alias_kind":"pith_short_16","alias_value":"UE3QTNKNC2LDAF6L","created_at":"2026-07-05T11:27:00.773651+00:00"},{"alias_kind":"pith_short_8","alias_value":"UE3QTNKN","created_at":"2026-07-05T11:27:00.773651+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09461","citing_title":"H2HMem: A Multimodal Memory Benchmark for Agents in Human-Human Interactions","ref_index":63,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UE3QTNKNC2LDAF6LMEDWSGIKEG","json":"https://pith.science/pith/UE3QTNKNC2LDAF6LMEDWSGIKEG.json","graph_json":"https://pith.science/api/pith-number/UE3QTNKNC2LDAF6LMEDWSGIKEG/graph.json","events_json":"https://pith.science/api/pith-number/UE3QTNKNC2LDAF6LMEDWSGIKEG/events.json","paper":"https://pith.science/paper/UE3QTNKN"},"agent_actions":{"view_html":"https://pith.science/pith/UE3QTNKNC2LDAF6LMEDWSGIKEG","download_json":"https://pith.science/pith/UE3QTNKNC2LDAF6LMEDWSGIKEG.json","view_paper":"https://pith.science/paper/UE3QTNKN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.20128&json=true","fetch_graph":"https://pith.science/api/pith-number/UE3QTNKNC2LDAF6LMEDWSGIKEG/graph.json","fetch_events":"https://pith.science/api/pith-number/UE3QTNKNC2LDAF6LMEDWSGIKEG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UE3QTNKNC2LDAF6LMEDWSGIKEG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UE3QTNKNC2LDAF6LMEDWSGIKEG/action/storage_attestation","attest_author":"https://pith.science/pith/UE3QTNKNC2LDAF6LMEDWSGIKEG/action/author_attestation","sign_citation":"https://pith.science/pith/UE3QTNKNC2LDAF6LMEDWSGIKEG/action/citation_signature","submit_replication":"https://pith.science/pith/UE3QTNKNC2LDAF6LMEDWSGIKEG/action/replication_record"}},"created_at":"2026-07-05T11:27:00.773651+00:00","updated_at":"2026-07-05T11:27:00.773651+00:00"}