{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BNAAHO7DUB5HGKEDNSMUUOU4PY","short_pith_number":"pith:BNAAHO7D","schema_version":"1.0","canonical_sha256":"0b4003bbe3a07a7328836c994a3a9c7e01bb040e6f8a030091d277d5f1510c27","source":{"kind":"arxiv","id":"2411.13212","version":3},"attestation_state":"computed","paper":{"title":"Limitations of Automatic Relevance Assessments with Large Language Models for Fair and Reliable Retrieval Evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.IR","authors_text":"\\'Alvaro Barreiro, David Otero, Javier Parapar","submitted_at":"2024-11-20T11:19:35Z","abstract_excerpt":"Offline evaluation of search systems depends on test collections. These benchmarks provide the researchers with a corpus of documents, topics and relevance judgements indicating which documents are relevant for each topic. While test collections are an integral part of Information Retrieval (IR) research, their creation involves significant efforts in manual annotation. Large language models (LLMs) are gaining much attention as tools for automatic relevance assessment. Recent research has shown that LLM-based assessments yield high systems ranking correlation with human-made judgements. These "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.13212","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.IR","submitted_at":"2024-11-20T11:19:35Z","cross_cats_sorted":[],"title_canon_sha256":"b1855dd873e522cb459442234d9e1d4341287d46f7a5d41f52d35f5660e5cbe7","abstract_canon_sha256":"89bb10371939ec4ed785b7fb0cae46d52f4a5b53c28c0e0679bdeb1c2b926cc9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:40:50.855689Z","signature_b64":"R1VTIPPxdDo9RXLLqgxRY+Emfmje4/uqOgFjIEwQqS9fxQXqnzE1DsjD3GekP+YE1qM9o5BSeg7DBpHGmtNTCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0b4003bbe3a07a7328836c994a3a9c7e01bb040e6f8a030091d277d5f1510c27","last_reissued_at":"2026-07-05T11:40:50.855204Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:40:50.855204Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Limitations of Automatic Relevance Assessments with Large Language Models for Fair and Reliable Retrieval Evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.IR","authors_text":"\\'Alvaro Barreiro, David Otero, Javier Parapar","submitted_at":"2024-11-20T11:19:35Z","abstract_excerpt":"Offline evaluation of search systems depends on test collections. These benchmarks provide the researchers with a corpus of documents, topics and relevance judgements indicating which documents are relevant for each topic. While test collections are an integral part of Information Retrieval (IR) research, their creation involves significant efforts in manual annotation. Large language models (LLMs) are gaining much attention as tools for automatic relevance assessment. Recent research has shown that LLM-based assessments yield high systems ranking correlation with human-made judgements. These "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.13212","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.13212/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.13212","created_at":"2026-07-05T11:40:50.855264+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.13212v3","created_at":"2026-07-05T11:40:50.855264+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.13212","created_at":"2026-07-05T11:40:50.855264+00:00"},{"alias_kind":"pith_short_12","alias_value":"BNAAHO7DUB5H","created_at":"2026-07-05T11:40:50.855264+00:00"},{"alias_kind":"pith_short_16","alias_value":"BNAAHO7DUB5HGKED","created_at":"2026-07-05T11:40:50.855264+00:00"},{"alias_kind":"pith_short_8","alias_value":"BNAAHO7D","created_at":"2026-07-05T11:40:50.855264+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BNAAHO7DUB5HGKEDNSMUUOU4PY","json":"https://pith.science/pith/BNAAHO7DUB5HGKEDNSMUUOU4PY.json","graph_json":"https://pith.science/api/pith-number/BNAAHO7DUB5HGKEDNSMUUOU4PY/graph.json","events_json":"https://pith.science/api/pith-number/BNAAHO7DUB5HGKEDNSMUUOU4PY/events.json","paper":"https://pith.science/paper/BNAAHO7D"},"agent_actions":{"view_html":"https://pith.science/pith/BNAAHO7DUB5HGKEDNSMUUOU4PY","download_json":"https://pith.science/pith/BNAAHO7DUB5HGKEDNSMUUOU4PY.json","view_paper":"https://pith.science/paper/BNAAHO7D","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.13212&json=true","fetch_graph":"https://pith.science/api/pith-number/BNAAHO7DUB5HGKEDNSMUUOU4PY/graph.json","fetch_events":"https://pith.science/api/pith-number/BNAAHO7DUB5HGKEDNSMUUOU4PY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BNAAHO7DUB5HGKEDNSMUUOU4PY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BNAAHO7DUB5HGKEDNSMUUOU4PY/action/storage_attestation","attest_author":"https://pith.science/pith/BNAAHO7DUB5HGKEDNSMUUOU4PY/action/author_attestation","sign_citation":"https://pith.science/pith/BNAAHO7DUB5HGKEDNSMUUOU4PY/action/citation_signature","submit_replication":"https://pith.science/pith/BNAAHO7DUB5HGKEDNSMUUOU4PY/action/replication_record"}},"created_at":"2026-07-05T11:40:50.855264+00:00","updated_at":"2026-07-05T11:40:50.855264+00:00"}