{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5HLWZZIPSFIUL7YZDJU2ABJWO6","short_pith_number":"pith:5HLWZZIP","schema_version":"1.0","canonical_sha256":"e9d76ce50f915145ff191a69a0053677ac4857490ee1ac1be690b6b2288b6712","source":{"kind":"arxiv","id":"2402.05129","version":1},"attestation_state":"computed","paper":{"title":"Best Practices for Text Annotation with Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Petter T\\\"ornberg","submitted_at":"2024-02-05T15:43:50Z","abstract_excerpt":"Large Language Models (LLMs) have ushered in a new era of text annotation, as their ease-of-use, high accuracy, and relatively low costs have meant that their use has exploded in recent months. However, the rapid growth of the field has meant that LLM-based annotation has become something of an academic Wild West: the lack of established practices and standards has led to concerns about the quality and validity of research. Researchers have warned that the ostensible simplicity of LLMs can be misleading, as they are prone to bias, misunderstandings, and unreliable results. Recognizing the tran"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.05129","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-02-05T15:43:50Z","cross_cats_sorted":[],"title_canon_sha256":"bbecc0687ebd248a4fb950a6a91e88b924c8ebeae0011271ecf539ee2acd20ad","abstract_canon_sha256":"2977d0df49670ef55cdce34f86432bf13cf84005f1700cffb1a676271bc0fb4a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:42:45.161667Z","signature_b64":"cyq9fvbc3BAT9/TFX5q5173rTVBsLZR/XDlAQQYLaA4BBR85H/huc3o7f7fJYnmO6sxd4IQ78XAn6TqOpTXxAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e9d76ce50f915145ff191a69a0053677ac4857490ee1ac1be690b6b2288b6712","last_reissued_at":"2026-07-05T07:42:45.161185Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:42:45.161185Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Best Practices for Text Annotation with Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Petter T\\\"ornberg","submitted_at":"2024-02-05T15:43:50Z","abstract_excerpt":"Large Language Models (LLMs) have ushered in a new era of text annotation, as their ease-of-use, high accuracy, and relatively low costs have meant that their use has exploded in recent months. However, the rapid growth of the field has meant that LLM-based annotation has become something of an academic Wild West: the lack of established practices and standards has led to concerns about the quality and validity of research. Researchers have warned that the ostensible simplicity of LLMs can be misleading, as they are prone to bias, misunderstandings, and unreliable results. Recognizing the tran"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.05129","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.05129/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.05129","created_at":"2026-07-05T07:42:45.161241+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.05129v1","created_at":"2026-07-05T07:42:45.161241+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.05129","created_at":"2026-07-05T07:42:45.161241+00:00"},{"alias_kind":"pith_short_12","alias_value":"5HLWZZIPSFIU","created_at":"2026-07-05T07:42:45.161241+00:00"},{"alias_kind":"pith_short_16","alias_value":"5HLWZZIPSFIUL7YZ","created_at":"2026-07-05T07:42:45.161241+00:00"},{"alias_kind":"pith_short_8","alias_value":"5HLWZZIP","created_at":"2026-07-05T07:42:45.161241+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.29393","citing_title":"The Role of Online Forums in Developer Understanding of Privacy Law -- A Reddit Case Study","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10786","citing_title":"Do BERT Embeddings Encode Narrative Dimensions? A Token-Level Probing Analysis of Time, Space, Causality, and Character in Fiction","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16765","citing_title":"Mapping Election Toxicity on Social Media across Issue, Ideology, and Psychosocial Dimensions","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5HLWZZIPSFIUL7YZDJU2ABJWO6","json":"https://pith.science/pith/5HLWZZIPSFIUL7YZDJU2ABJWO6.json","graph_json":"https://pith.science/api/pith-number/5HLWZZIPSFIUL7YZDJU2ABJWO6/graph.json","events_json":"https://pith.science/api/pith-number/5HLWZZIPSFIUL7YZDJU2ABJWO6/events.json","paper":"https://pith.science/paper/5HLWZZIP"},"agent_actions":{"view_html":"https://pith.science/pith/5HLWZZIPSFIUL7YZDJU2ABJWO6","download_json":"https://pith.science/pith/5HLWZZIPSFIUL7YZDJU2ABJWO6.json","view_paper":"https://pith.science/paper/5HLWZZIP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.05129&json=true","fetch_graph":"https://pith.science/api/pith-number/5HLWZZIPSFIUL7YZDJU2ABJWO6/graph.json","fetch_events":"https://pith.science/api/pith-number/5HLWZZIPSFIUL7YZDJU2ABJWO6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5HLWZZIPSFIUL7YZDJU2ABJWO6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5HLWZZIPSFIUL7YZDJU2ABJWO6/action/storage_attestation","attest_author":"https://pith.science/pith/5HLWZZIPSFIUL7YZDJU2ABJWO6/action/author_attestation","sign_citation":"https://pith.science/pith/5HLWZZIPSFIUL7YZDJU2ABJWO6/action/citation_signature","submit_replication":"https://pith.science/pith/5HLWZZIPSFIUL7YZDJU2ABJWO6/action/replication_record"}},"created_at":"2026-07-05T07:42:45.161241+00:00","updated_at":"2026-07-05T07:42:45.161241+00:00"}