{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:TWIGRFN7NLJELXP5VVEG3LRH4Z","short_pith_number":"pith:TWIGRFN7","schema_version":"1.0","canonical_sha256":"9d906895bf6ad245ddfdad486dae27e64afca856a8b09f3d3035ca93a273a68f","source":{"kind":"arxiv","id":"2304.06588","version":1},"attestation_state":"computed","paper":{"title":"ChatGPT-4 Outperforms Experts and Crowd Workers in Annotating Political Twitter Messages with Zero-Shot Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.SI"],"primary_cat":"cs.CL","authors_text":"Petter T\\\"ornberg","submitted_at":"2023-04-13T14:51:40Z","abstract_excerpt":"This paper assesses the accuracy, reliability and bias of the Large Language Model (LLM) ChatGPT-4 on the text analysis task of classifying the political affiliation of a Twitter poster based on the content of a tweet. The LLM is compared to manual annotation by both expert classifiers and crowd workers, generally considered the gold standard for such tasks. We use Twitter messages from United States politicians during the 2020 election, providing a ground truth against which to measure accuracy. The paper finds that ChatGPT-4 has achieves higher accuracy, higher reliability, and equal or lowe"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.06588","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-04-13T14:51:40Z","cross_cats_sorted":["cs.AI","cs.SI"],"title_canon_sha256":"ea3a5a68e671f9e66d793a6573b558e17029c0115a28eceb8c52cff6e638794d","abstract_canon_sha256":"6a1fa088771bb404a3e4e1b9538f53b26e14ac767722109f2789bb8764f9788a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:00:45.440094Z","signature_b64":"DXQhN0N/MepHbWLaK5TkWlY3NVCLhh557ff0nCKlEBIfR+M1Q3QyA2Jfof30GAbgxNGHzwjSctJ84tWbo+yGDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9d906895bf6ad245ddfdad486dae27e64afca856a8b09f3d3035ca93a273a68f","last_reissued_at":"2026-07-05T06:00:45.439677Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:00:45.439677Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ChatGPT-4 Outperforms Experts and Crowd Workers in Annotating Political Twitter Messages with Zero-Shot Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.SI"],"primary_cat":"cs.CL","authors_text":"Petter T\\\"ornberg","submitted_at":"2023-04-13T14:51:40Z","abstract_excerpt":"This paper assesses the accuracy, reliability and bias of the Large Language Model (LLM) ChatGPT-4 on the text analysis task of classifying the political affiliation of a Twitter poster based on the content of a tweet. The LLM is compared to manual annotation by both expert classifiers and crowd workers, generally considered the gold standard for such tasks. We use Twitter messages from United States politicians during the 2020 election, providing a ground truth against which to measure accuracy. The paper finds that ChatGPT-4 has achieves higher accuracy, higher reliability, and equal or lowe"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.06588","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.06588/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.06588","created_at":"2026-07-05T06:00:45.439733+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.06588v1","created_at":"2026-07-05T06:00:45.439733+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.06588","created_at":"2026-07-05T06:00:45.439733+00:00"},{"alias_kind":"pith_short_12","alias_value":"TWIGRFN7NLJE","created_at":"2026-07-05T06:00:45.439733+00:00"},{"alias_kind":"pith_short_16","alias_value":"TWIGRFN7NLJELXP5","created_at":"2026-07-05T06:00:45.439733+00:00"},{"alias_kind":"pith_short_8","alias_value":"TWIGRFN7","created_at":"2026-07-05T06:00:45.439733+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26749","citing_title":"Structure Before Collapse: Transient semantic geometry in next-token prediction","ref_index":99,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23042","citing_title":"The Model as One Rater Among Several: Measuring Political Positions in Data-Sparse Regions with a Language-Model Panel","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18889","citing_title":"Improving Medical Communication using Rubric-Guided Counterfactual Recommendations","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17503","citing_title":"What Prediction Markets Can See: Market Formation, Settlement Legibility, and the Geography of Tradable Uncertainty in Africa and Latin America","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30801","citing_title":"Using AI Agents to Automate Black-Box Audits of Personalization Algorithms at Scale","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04273","citing_title":"Characterizing initial human-AI proof formalization workflows","ref_index":253,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20693","citing_title":"Interpretable Discriminative Text Representations via Agreement and Label Disentanglement","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2506.21582","citing_title":"VIDEE: Visual and Interactive Decomposition, Execution, and Evaluation of Text Analytics with Intelligent Agents","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2509.06337","citing_title":"Large Language Models as Virtual Survey Respondents: Evaluating Sociodemographic Response Generation","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15329","citing_title":"Evaluating LLMs as Human Surrogates in Controlled Experiments","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05579","citing_title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","ref_index":225,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18955","citing_title":"Assessing Capabilities of Large Language Models in Social Media Analytics: A Multi-task Quest","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07530","citing_title":"The Shrinking Lifespan of LLMs in Science","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14672","citing_title":"SPAGBias: Uncovering and Tracing Structured Spatial Gender Bias in Large Language Models","ref_index":59,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TWIGRFN7NLJELXP5VVEG3LRH4Z","json":"https://pith.science/pith/TWIGRFN7NLJELXP5VVEG3LRH4Z.json","graph_json":"https://pith.science/api/pith-number/TWIGRFN7NLJELXP5VVEG3LRH4Z/graph.json","events_json":"https://pith.science/api/pith-number/TWIGRFN7NLJELXP5VVEG3LRH4Z/events.json","paper":"https://pith.science/paper/TWIGRFN7"},"agent_actions":{"view_html":"https://pith.science/pith/TWIGRFN7NLJELXP5VVEG3LRH4Z","download_json":"https://pith.science/pith/TWIGRFN7NLJELXP5VVEG3LRH4Z.json","view_paper":"https://pith.science/paper/TWIGRFN7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.06588&json=true","fetch_graph":"https://pith.science/api/pith-number/TWIGRFN7NLJELXP5VVEG3LRH4Z/graph.json","fetch_events":"https://pith.science/api/pith-number/TWIGRFN7NLJELXP5VVEG3LRH4Z/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TWIGRFN7NLJELXP5VVEG3LRH4Z/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TWIGRFN7NLJELXP5VVEG3LRH4Z/action/storage_attestation","attest_author":"https://pith.science/pith/TWIGRFN7NLJELXP5VVEG3LRH4Z/action/author_attestation","sign_citation":"https://pith.science/pith/TWIGRFN7NLJELXP5VVEG3LRH4Z/action/citation_signature","submit_replication":"https://pith.science/pith/TWIGRFN7NLJELXP5VVEG3LRH4Z/action/replication_record"}},"created_at":"2026-07-05T06:00:45.439733+00:00","updated_at":"2026-07-05T06:00:45.439733+00:00"}