{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:BOJYVLUT3UXFOT3IUR6K3BVFHM","short_pith_number":"pith:BOJYVLUT","schema_version":"1.0","canonical_sha256":"0b938aae93dd2e574f68a47cad86a53b1ff179a0f60b0c0402b4db1b5189a36e","source":{"kind":"arxiv","id":"2501.10970","version":4},"attestation_state":"computed","paper":{"title":"The Alternative Annotator Test for LLM-as-a-Judge: How to Statistically Justify Replacing Human Annotators with LLMs","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.AI","cs.HC"],"primary_cat":"cs.CL","authors_text":"Nitay Calderon, Roi Reichart, Rotem Dror","submitted_at":"2025-01-19T07:09:11Z","abstract_excerpt":"The \"LLM-as-an-annotator\" and \"LLM-as-a-judge\" paradigms employ Large Language Models (LLMs) as annotators, judges, and evaluators in tasks traditionally performed by humans. LLM annotations are widely used, not only in NLP research but also in fields like medicine, psychology, and social science. Despite their role in shaping study results and insights, there is no standard or rigorous procedure to determine whether LLMs can replace human annotators. In this paper, we propose a novel statistical procedure, the Alternative Annotator Test (alt-test), that requires only a modest subset of annota"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.10970","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.CL","submitted_at":"2025-01-19T07:09:11Z","cross_cats_sorted":["cs.AI","cs.HC"],"title_canon_sha256":"db28c71b89957904bad69fe9910c31de36af51fb91351deda875b86f68c8f2c7","abstract_canon_sha256":"245b81a42ae9753dafbf34725ad5c71378a690667c17a43f6371a7d076f98238"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:50:30.109236Z","signature_b64":"1WdKb5PHsHToIhoWpT29Cz+cudW8KN+OU/g8Qnk1EH6mXYI1k92Kg5optHji/RvsKvrhkSIxDmOptZryvsKMDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0b938aae93dd2e574f68a47cad86a53b1ff179a0f60b0c0402b4db1b5189a36e","last_reissued_at":"2026-07-05T11:50:30.108754Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:50:30.108754Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Alternative Annotator Test for LLM-as-a-Judge: How to Statistically Justify Replacing Human Annotators with LLMs","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.AI","cs.HC"],"primary_cat":"cs.CL","authors_text":"Nitay Calderon, Roi Reichart, Rotem Dror","submitted_at":"2025-01-19T07:09:11Z","abstract_excerpt":"The \"LLM-as-an-annotator\" and \"LLM-as-a-judge\" paradigms employ Large Language Models (LLMs) as annotators, judges, and evaluators in tasks traditionally performed by humans. LLM annotations are widely used, not only in NLP research but also in fields like medicine, psychology, and social science. Despite their role in shaping study results and insights, there is no standard or rigorous procedure to determine whether LLMs can replace human annotators. In this paper, we propose a novel statistical procedure, the Alternative Annotator Test (alt-test), that requires only a modest subset of annota"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.10970","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.10970/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.10970","created_at":"2026-07-05T11:50:30.108814+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.10970v4","created_at":"2026-07-05T11:50:30.108814+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.10970","created_at":"2026-07-05T11:50:30.108814+00:00"},{"alias_kind":"pith_short_12","alias_value":"BOJYVLUT3UXF","created_at":"2026-07-05T11:50:30.108814+00:00"},{"alias_kind":"pith_short_16","alias_value":"BOJYVLUT3UXFOT3I","created_at":"2026-07-05T11:50:30.108814+00:00"},{"alias_kind":"pith_short_8","alias_value":"BOJYVLUT","created_at":"2026-07-05T11:50:30.108814+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.27025","citing_title":"Attribute-Based Diagnosis of LLM Alignment with Hate Speech Annotations","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2502.17773","citing_title":"How Many Human Survey Respondents is a Large Language Model Worth? An Uncertainty Quantification Perspective","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15365","citing_title":"Greedy or not, here I come: Language production under vocabulary constraints in humans and resource-rational models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2507.05385","citing_title":"EduCoder: An Open-Source Annotation System for Education Transcript Data","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BOJYVLUT3UXFOT3IUR6K3BVFHM","json":"https://pith.science/pith/BOJYVLUT3UXFOT3IUR6K3BVFHM.json","graph_json":"https://pith.science/api/pith-number/BOJYVLUT3UXFOT3IUR6K3BVFHM/graph.json","events_json":"https://pith.science/api/pith-number/BOJYVLUT3UXFOT3IUR6K3BVFHM/events.json","paper":"https://pith.science/paper/BOJYVLUT"},"agent_actions":{"view_html":"https://pith.science/pith/BOJYVLUT3UXFOT3IUR6K3BVFHM","download_json":"https://pith.science/pith/BOJYVLUT3UXFOT3IUR6K3BVFHM.json","view_paper":"https://pith.science/paper/BOJYVLUT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.10970&json=true","fetch_graph":"https://pith.science/api/pith-number/BOJYVLUT3UXFOT3IUR6K3BVFHM/graph.json","fetch_events":"https://pith.science/api/pith-number/BOJYVLUT3UXFOT3IUR6K3BVFHM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BOJYVLUT3UXFOT3IUR6K3BVFHM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BOJYVLUT3UXFOT3IUR6K3BVFHM/action/storage_attestation","attest_author":"https://pith.science/pith/BOJYVLUT3UXFOT3IUR6K3BVFHM/action/author_attestation","sign_citation":"https://pith.science/pith/BOJYVLUT3UXFOT3IUR6K3BVFHM/action/citation_signature","submit_replication":"https://pith.science/pith/BOJYVLUT3UXFOT3IUR6K3BVFHM/action/replication_record"}},"created_at":"2026-07-05T11:50:30.108814+00:00","updated_at":"2026-07-05T11:50:30.108814+00:00"}