{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:R7F4HWJNP3IU4AB4Q7ZNRDTFD5","short_pith_number":"pith:R7F4HWJN","schema_version":"1.0","canonical_sha256":"8fcbc3d92d7ed14e003c87f2d88e651f4e946adcc01a6015552f2d79e7359707","source":{"kind":"arxiv","id":"2304.11085","version":1},"attestation_state":"computed","paper":{"title":"Testing the Reliability of ChatGPT for Text Annotation and Classification: A Cautionary Remark","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Michael V. Reiss","submitted_at":"2023-04-17T00:41:19Z","abstract_excerpt":"Recent studies have demonstrated promising potential of ChatGPT for various text annotation and classification tasks. However, ChatGPT is non-deterministic which means that, as with human coders, identical input can lead to different outputs. Given this, it seems appropriate to test the reliability of ChatGPT. Therefore, this study investigates the consistency of ChatGPT's zero-shot capabilities for text annotation and classification, focusing on different model parameters, prompt variations, and repetitions of identical inputs. Based on the real-world classification task of differentiating we"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.11085","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2023-04-17T00:41:19Z","cross_cats_sorted":[],"title_canon_sha256":"45f12afe7b2618265224648e303ae1c773130ce6726dd51e5dfbb4e0cc15eda6","abstract_canon_sha256":"2bae3752b33fb5273cc1fa1b7152b5000802650d7d4f3d3211fd25475235c578"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:03:18.881835Z","signature_b64":"OY+WZY/c2mosML8kpIVHyvqyqwHc18y2aXbykxkyp+whJuUfjtDhgnvswr0y0jUG32XUT2jf+FTC5F7CWRbrCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8fcbc3d92d7ed14e003c87f2d88e651f4e946adcc01a6015552f2d79e7359707","last_reissued_at":"2026-07-05T06:03:18.881432Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:03:18.881432Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Testing the Reliability of ChatGPT for Text Annotation and Classification: A Cautionary Remark","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Michael V. Reiss","submitted_at":"2023-04-17T00:41:19Z","abstract_excerpt":"Recent studies have demonstrated promising potential of ChatGPT for various text annotation and classification tasks. However, ChatGPT is non-deterministic which means that, as with human coders, identical input can lead to different outputs. Given this, it seems appropriate to test the reliability of ChatGPT. Therefore, this study investigates the consistency of ChatGPT's zero-shot capabilities for text annotation and classification, focusing on different model parameters, prompt variations, and repetitions of identical inputs. Based on the real-world classification task of differentiating we"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.11085","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.11085/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.11085","created_at":"2026-07-05T06:03:18.881488+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.11085v1","created_at":"2026-07-05T06:03:18.881488+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.11085","created_at":"2026-07-05T06:03:18.881488+00:00"},{"alias_kind":"pith_short_12","alias_value":"R7F4HWJNP3IU","created_at":"2026-07-05T06:03:18.881488+00:00"},{"alias_kind":"pith_short_16","alias_value":"R7F4HWJNP3IU4AB4","created_at":"2026-07-05T06:03:18.881488+00:00"},{"alias_kind":"pith_short_8","alias_value":"R7F4HWJN","created_at":"2026-07-05T06:03:18.881488+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11699","citing_title":"A Data-Centric Framework for Detecting and Correcting Corrupted Labels","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11695","citing_title":"Noise-Aware Framework for Correcting Corrupted Labels","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2508.15503","citing_title":"Guidelines for Empirical Studies in Software Engineering involving Large Language Models","ref_index":108,"is_internal_anchor":false},{"citing_arxiv_id":"2508.15503","citing_title":"Guidelines for Empirical Studies in Software Engineering involving Large Language Models","ref_index":108,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/R7F4HWJNP3IU4AB4Q7ZNRDTFD5","json":"https://pith.science/pith/R7F4HWJNP3IU4AB4Q7ZNRDTFD5.json","graph_json":"https://pith.science/api/pith-number/R7F4HWJNP3IU4AB4Q7ZNRDTFD5/graph.json","events_json":"https://pith.science/api/pith-number/R7F4HWJNP3IU4AB4Q7ZNRDTFD5/events.json","paper":"https://pith.science/paper/R7F4HWJN"},"agent_actions":{"view_html":"https://pith.science/pith/R7F4HWJNP3IU4AB4Q7ZNRDTFD5","download_json":"https://pith.science/pith/R7F4HWJNP3IU4AB4Q7ZNRDTFD5.json","view_paper":"https://pith.science/paper/R7F4HWJN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.11085&json=true","fetch_graph":"https://pith.science/api/pith-number/R7F4HWJNP3IU4AB4Q7ZNRDTFD5/graph.json","fetch_events":"https://pith.science/api/pith-number/R7F4HWJNP3IU4AB4Q7ZNRDTFD5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/R7F4HWJNP3IU4AB4Q7ZNRDTFD5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/R7F4HWJNP3IU4AB4Q7ZNRDTFD5/action/storage_attestation","attest_author":"https://pith.science/pith/R7F4HWJNP3IU4AB4Q7ZNRDTFD5/action/author_attestation","sign_citation":"https://pith.science/pith/R7F4HWJNP3IU4AB4Q7ZNRDTFD5/action/citation_signature","submit_replication":"https://pith.science/pith/R7F4HWJNP3IU4AB4Q7ZNRDTFD5/action/replication_record"}},"created_at":"2026-07-05T06:03:18.881488+00:00","updated_at":"2026-07-05T06:03:18.881488+00:00"}