{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:SWXFQ3FD4CO4KDQIFS2RIPKA6G","short_pith_number":"pith:SWXFQ3FD","schema_version":"1.0","canonical_sha256":"95ae586ca3e09dc50e082cb5143d40f1a9e5c9fcc99a935f526c77ae490dce8a","source":{"kind":"arxiv","id":"2502.20937","version":2},"attestation_state":"computed","paper":{"title":"Variations in Relevance Judgments and the Shelf Life of Test Collections","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.IR","authors_text":"Andrew Parry, Eugene Yang, Ferdinand Schlatt, Guglielmo Faggioli, Harrisen Scells, Maik Fr\\\"obe, Saber Zerhoudi, Sean MacAvaney","submitted_at":"2025-02-28T10:46:56Z","abstract_excerpt":"The fundamental property of Cranfield-style evaluations, that system rankings are stable even when assessors disagree on individual relevance decisions, was validated on traditional test collections. However, the paradigm shift towards neural retrieval models affected the characteristics of modern test collections, e.g., documents are short, judged with four grades of relevance, and information needs have no descriptions or narratives. Under these changes, it is unclear whether assessor disagreement remains negligible for system comparisons. We investigate this aspect under the additional cond"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.20937","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.IR","submitted_at":"2025-02-28T10:46:56Z","cross_cats_sorted":[],"title_canon_sha256":"3a3b194f58c019540314aab84d4e146596f5a33e73a1a31fbda98396ed904eec","abstract_canon_sha256":"88b512b1a8598e0d00a69aa4d41d6ac0dfe7517e62fe3879bff488f7c54e67c6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:06:28.995892Z","signature_b64":"qZYxJbgofBdUychL5rNySvrdTHtzsRG6oX42kiRhVA+xq9FYX7MCSu7c75AZQbTHKoRY8LfG4SUO3H+Ag11fBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"95ae586ca3e09dc50e082cb5143d40f1a9e5c9fcc99a935f526c77ae490dce8a","last_reissued_at":"2026-07-05T11:06:28.995287Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:06:28.995287Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Variations in Relevance Judgments and the Shelf Life of Test Collections","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.IR","authors_text":"Andrew Parry, Eugene Yang, Ferdinand Schlatt, Guglielmo Faggioli, Harrisen Scells, Maik Fr\\\"obe, Saber Zerhoudi, Sean MacAvaney","submitted_at":"2025-02-28T10:46:56Z","abstract_excerpt":"The fundamental property of Cranfield-style evaluations, that system rankings are stable even when assessors disagree on individual relevance decisions, was validated on traditional test collections. However, the paradigm shift towards neural retrieval models affected the characteristics of modern test collections, e.g., documents are short, judged with four grades of relevance, and information needs have no descriptions or narratives. Under these changes, it is unclear whether assessor disagreement remains negligible for system comparisons. We investigate this aspect under the additional cond"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.20937","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.20937/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.20937","created_at":"2026-07-05T11:06:28.995363+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.20937v2","created_at":"2026-07-05T11:06:28.995363+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.20937","created_at":"2026-07-05T11:06:28.995363+00:00"},{"alias_kind":"pith_short_12","alias_value":"SWXFQ3FD4CO4","created_at":"2026-07-05T11:06:28.995363+00:00"},{"alias_kind":"pith_short_16","alias_value":"SWXFQ3FD4CO4KDQI","created_at":"2026-07-05T11:06:28.995363+00:00"},{"alias_kind":"pith_short_8","alias_value":"SWXFQ3FD","created_at":"2026-07-05T11:06:28.995363+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.04013","citing_title":"On Robustness and Reliability of Benchmark-Based Evaluation of LLMs","ref_index":29,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SWXFQ3FD4CO4KDQIFS2RIPKA6G","json":"https://pith.science/pith/SWXFQ3FD4CO4KDQIFS2RIPKA6G.json","graph_json":"https://pith.science/api/pith-number/SWXFQ3FD4CO4KDQIFS2RIPKA6G/graph.json","events_json":"https://pith.science/api/pith-number/SWXFQ3FD4CO4KDQIFS2RIPKA6G/events.json","paper":"https://pith.science/paper/SWXFQ3FD"},"agent_actions":{"view_html":"https://pith.science/pith/SWXFQ3FD4CO4KDQIFS2RIPKA6G","download_json":"https://pith.science/pith/SWXFQ3FD4CO4KDQIFS2RIPKA6G.json","view_paper":"https://pith.science/paper/SWXFQ3FD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.20937&json=true","fetch_graph":"https://pith.science/api/pith-number/SWXFQ3FD4CO4KDQIFS2RIPKA6G/graph.json","fetch_events":"https://pith.science/api/pith-number/SWXFQ3FD4CO4KDQIFS2RIPKA6G/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SWXFQ3FD4CO4KDQIFS2RIPKA6G/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SWXFQ3FD4CO4KDQIFS2RIPKA6G/action/storage_attestation","attest_author":"https://pith.science/pith/SWXFQ3FD4CO4KDQIFS2RIPKA6G/action/author_attestation","sign_citation":"https://pith.science/pith/SWXFQ3FD4CO4KDQIFS2RIPKA6G/action/citation_signature","submit_replication":"https://pith.science/pith/SWXFQ3FD4CO4KDQIFS2RIPKA6G/action/replication_record"}},"created_at":"2026-07-05T11:06:28.995363+00:00","updated_at":"2026-07-05T11:06:28.995363+00:00"}