{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:AO76HISHDTITHZQOZOZVTCQUOP","short_pith_number":"pith:AO76HISH","schema_version":"1.0","canonical_sha256":"03bfe3a2471cd133e60ecbb3598a1473d12116ea184a2bfaec4c944500bb9e46","source":{"kind":"arxiv","id":"2405.05600","version":1},"attestation_state":"computed","paper":{"title":"Can We Use Large Language Models to Fill Relevance Judgment Holes?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.IR","authors_text":"Chuan Meng, Leif Azzopardi, Mohammad Aliannejadi, Zahra Abbasiantaeb","submitted_at":"2024-05-09T07:39:19Z","abstract_excerpt":"Incomplete relevance judgments limit the re-usability of test collections. When new systems are compared against previous systems used to build the pool of judged documents, they often do so at a disadvantage due to the ``holes'' in test collection (i.e., pockets of un-assessed documents returned by the new system). In this paper, we take initial steps towards extending existing test collections by employing Large Language Models (LLM) to fill the holes by leveraging and grounding the method using existing human judgments. We explore this problem in the context of Conversational Search using T"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.05600","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.IR","submitted_at":"2024-05-09T07:39:19Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"62316c9daf4568d624893f148a0b32641adb045ca23a72581fa7cdfd460ae1b3","abstract_canon_sha256":"5bb64748e8afa0d8ac67693ae6d98418935d533631dbf3925f32aca8779cdb4f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:17:20.407441Z","signature_b64":"kCyJf6Ea9PaE5Nwej2zFFYpVoeDZpDhfL+hEDatYfCwVz4yhTDeAONJxFOJU4IJwwt0kSCcczG83+003psMGCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"03bfe3a2471cd133e60ecbb3598a1473d12116ea184a2bfaec4c944500bb9e46","last_reissued_at":"2026-07-05T08:17:20.407023Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:17:20.407023Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Can We Use Large Language Models to Fill Relevance Judgment Holes?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.IR","authors_text":"Chuan Meng, Leif Azzopardi, Mohammad Aliannejadi, Zahra Abbasiantaeb","submitted_at":"2024-05-09T07:39:19Z","abstract_excerpt":"Incomplete relevance judgments limit the re-usability of test collections. When new systems are compared against previous systems used to build the pool of judged documents, they often do so at a disadvantage due to the ``holes'' in test collection (i.e., pockets of un-assessed documents returned by the new system). In this paper, we take initial steps towards extending existing test collections by employing Large Language Models (LLM) to fill the holes by leveraging and grounding the method using existing human judgments. We explore this problem in the context of Conversational Search using T"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.05600","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.05600/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.05600","created_at":"2026-07-05T08:17:20.407084+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.05600v1","created_at":"2026-07-05T08:17:20.407084+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.05600","created_at":"2026-07-05T08:17:20.407084+00:00"},{"alias_kind":"pith_short_12","alias_value":"AO76HISHDTIT","created_at":"2026-07-05T08:17:20.407084+00:00"},{"alias_kind":"pith_short_16","alias_value":"AO76HISHDTITHZQO","created_at":"2026-07-05T08:17:20.407084+00:00"},{"alias_kind":"pith_short_8","alias_value":"AO76HISH","created_at":"2026-07-05T08:17:20.407084+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2601.08919","citing_title":"LLMs as Assessors: Right for the Right Reason?","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2602.08457","citing_title":"Hybrid Pooling with LLMs via Relevance Context Learning","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AO76HISHDTITHZQOZOZVTCQUOP","json":"https://pith.science/pith/AO76HISHDTITHZQOZOZVTCQUOP.json","graph_json":"https://pith.science/api/pith-number/AO76HISHDTITHZQOZOZVTCQUOP/graph.json","events_json":"https://pith.science/api/pith-number/AO76HISHDTITHZQOZOZVTCQUOP/events.json","paper":"https://pith.science/paper/AO76HISH"},"agent_actions":{"view_html":"https://pith.science/pith/AO76HISHDTITHZQOZOZVTCQUOP","download_json":"https://pith.science/pith/AO76HISHDTITHZQOZOZVTCQUOP.json","view_paper":"https://pith.science/paper/AO76HISH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.05600&json=true","fetch_graph":"https://pith.science/api/pith-number/AO76HISHDTITHZQOZOZVTCQUOP/graph.json","fetch_events":"https://pith.science/api/pith-number/AO76HISHDTITHZQOZOZVTCQUOP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AO76HISHDTITHZQOZOZVTCQUOP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AO76HISHDTITHZQOZOZVTCQUOP/action/storage_attestation","attest_author":"https://pith.science/pith/AO76HISHDTITHZQOZOZVTCQUOP/action/author_attestation","sign_citation":"https://pith.science/pith/AO76HISHDTITHZQOZOZVTCQUOP/action/citation_signature","submit_replication":"https://pith.science/pith/AO76HISHDTITHZQOZOZVTCQUOP/action/replication_record"}},"created_at":"2026-07-05T08:17:20.407084+00:00","updated_at":"2026-07-05T08:17:20.407084+00:00"}