{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:WI6OAYJFOARK65U3Z7CTN62TW7","short_pith_number":"pith:WI6OAYJF","schema_version":"1.0","canonical_sha256":"b23ce061257022af769bcfc536fb53b7dba3b0e383e137f8829b0e5f59444132","source":{"kind":"arxiv","id":"2306.00176","version":1},"attestation_state":"computed","paper":{"title":"Automated Annotation with Generative AI Requires Validation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Neil Fasching, Nicholas Pangakis, Samuel Wolken","submitted_at":"2023-05-31T20:50:45Z","abstract_excerpt":"Generative large language models (LLMs) can be a powerful tool for augmenting text annotation procedures, but their performance varies across annotation tasks due to prompt quality, text data idiosyncrasies, and conceptual difficulty. Because these challenges will persist even as LLM technology improves, we argue that any automated annotation process using an LLM must validate the LLM's performance against labels generated by humans. To this end, we outline a workflow to harness the annotation potential of LLMs in a principled, efficient way. Using GPT-4, we validate this approach by replicati"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.00176","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-05-31T20:50:45Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"0541c048c2a673be3da6649dc9ec8fe62c85bc0690f3908f7c207b0aa3692064","abstract_canon_sha256":"e1657b1a4eef428fdd2bc766295476545766184d6266a8de982b8ce3873a6551"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:16:23.031662Z","signature_b64":"HgDDq4lG7/dPRv2juLKdS9ClhhPUlFYNdxuAnVHFR2RTjsZXUow+lTpJNWqJ5748etJd8DITxNC5cwg8dK+TBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b23ce061257022af769bcfc536fb53b7dba3b0e383e137f8829b0e5f59444132","last_reissued_at":"2026-07-05T06:16:23.031235Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:16:23.031235Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Automated Annotation with Generative AI Requires Validation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Neil Fasching, Nicholas Pangakis, Samuel Wolken","submitted_at":"2023-05-31T20:50:45Z","abstract_excerpt":"Generative large language models (LLMs) can be a powerful tool for augmenting text annotation procedures, but their performance varies across annotation tasks due to prompt quality, text data idiosyncrasies, and conceptual difficulty. Because these challenges will persist even as LLM technology improves, we argue that any automated annotation process using an LLM must validate the LLM's performance against labels generated by humans. To this end, we outline a workflow to harness the annotation potential of LLMs in a principled, efficient way. Using GPT-4, we validate this approach by replicati"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.00176","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.00176/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.00176","created_at":"2026-07-05T06:16:23.031297+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.00176v1","created_at":"2026-07-05T06:16:23.031297+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.00176","created_at":"2026-07-05T06:16:23.031297+00:00"},{"alias_kind":"pith_short_12","alias_value":"WI6OAYJFOARK","created_at":"2026-07-05T06:16:23.031297+00:00"},{"alias_kind":"pith_short_16","alias_value":"WI6OAYJFOARK65U3","created_at":"2026-07-05T06:16:23.031297+00:00"},{"alias_kind":"pith_short_8","alias_value":"WI6OAYJF","created_at":"2026-07-05T06:16:23.031297+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08346","citing_title":"Grounded Event Extraction from SEC 8-K Filings with a Fine-Grained Taxonomy","ref_index":18,"is_internal_anchor":true},{"citing_arxiv_id":"2606.23042","citing_title":"The Model as One Rater Among Several: Measuring Political Positions in Data-Sparse Regions with a Language-Model Panel","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06784","citing_title":"What Your Posts Reveal: A Benchmark and Agentic Framework for User-Level Privacy Leakage on Social Media","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00551","citing_title":"Talking Politics with Artificial Intelligence","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28574","citing_title":"Correct codes for the wrong reasons? validating LLMs as measurement instruments for theoretical constructs","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2508.15503","citing_title":"Guidelines for Empirical Studies in Software Engineering involving Large Language Models","ref_index":100,"is_internal_anchor":false},{"citing_arxiv_id":"2508.15503","citing_title":"Guidelines for Empirical Studies in Software Engineering involving Large Language Models","ref_index":100,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19783","citing_title":"How Much Does Persuasion Strategy Matter? LLM-Annotated Evidence from Charitable Donation Dialogues","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18878","citing_title":"LegalBench-BR: A Benchmark for Evaluating Large Language Models on Brazilian Legal Decision Classification","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14321","citing_title":"LLM Predictive Scoring and Validation: Inferring Experience Ratings from Unstructured Text","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13899","citing_title":"Do We Still Need Humans in the Loop? Comparing Human and LLM Annotation in Active Learning for Hostility Detection","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WI6OAYJFOARK65U3Z7CTN62TW7","json":"https://pith.science/pith/WI6OAYJFOARK65U3Z7CTN62TW7.json","graph_json":"https://pith.science/api/pith-number/WI6OAYJFOARK65U3Z7CTN62TW7/graph.json","events_json":"https://pith.science/api/pith-number/WI6OAYJFOARK65U3Z7CTN62TW7/events.json","paper":"https://pith.science/paper/WI6OAYJF"},"agent_actions":{"view_html":"https://pith.science/pith/WI6OAYJFOARK65U3Z7CTN62TW7","download_json":"https://pith.science/pith/WI6OAYJFOARK65U3Z7CTN62TW7.json","view_paper":"https://pith.science/paper/WI6OAYJF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.00176&json=true","fetch_graph":"https://pith.science/api/pith-number/WI6OAYJFOARK65U3Z7CTN62TW7/graph.json","fetch_events":"https://pith.science/api/pith-number/WI6OAYJFOARK65U3Z7CTN62TW7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WI6OAYJFOARK65U3Z7CTN62TW7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WI6OAYJFOARK65U3Z7CTN62TW7/action/storage_attestation","attest_author":"https://pith.science/pith/WI6OAYJFOARK65U3Z7CTN62TW7/action/author_attestation","sign_citation":"https://pith.science/pith/WI6OAYJFOARK65U3Z7CTN62TW7/action/citation_signature","submit_replication":"https://pith.science/pith/WI6OAYJFOARK65U3Z7CTN62TW7/action/replication_record"}},"created_at":"2026-07-05T06:16:23.031297+00:00","updated_at":"2026-07-05T06:16:23.031297+00:00"}