{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:33PWX6OEEOMVLL4WHOGPINU6DK","short_pith_number":"pith:33PWX6OE","schema_version":"1.0","canonical_sha256":"dedf6bf9c4239955af963b8cf4369e1a891068eb13ab14f94b52d048373652fe","source":{"kind":"arxiv","id":"2306.02177","version":1},"attestation_state":"computed","paper":{"title":"Towards Coding Social Science Datasets with Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Christopher Michael Rytting, David Wingate, Ethan Busby, Joshua Gubler, Lisa Argyle, Nancy Fulda, Taylor Sorensen","submitted_at":"2023-06-03T19:11:34Z","abstract_excerpt":"Researchers often rely on humans to code (label, annotate, etc.) large sets of texts. This kind of human coding forms an important part of social science research, yet the coding process is both resource intensive and highly variable from application to application. In some cases, efforts to automate this process have achieved human-level accuracies, but to achieve this, these attempts frequently rely on thousands of hand-labeled training examples, which makes them inapplicable to small-scale research studies and costly for large ones. Recent advances in a specific kind of artificial intellige"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.02177","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2023-06-03T19:11:34Z","cross_cats_sorted":[],"title_canon_sha256":"188fa3352aea3db67f4f75dab60639f0a03a4fc0c78ebec1681760d5de37bf70","abstract_canon_sha256":"c8abf925e3c34b6f6ccef043c3351d747e2db8490a43377a7643be98a22eac3f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:17:13.897403Z","signature_b64":"sSijEGRwyLVPQUl1iDPgvciRoXAmpC9U7wdmkdPUJblVktsQLF6rIWGqzfIXHqKhNoPUrMyU8/XQypxgc388AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dedf6bf9c4239955af963b8cf4369e1a891068eb13ab14f94b52d048373652fe","last_reissued_at":"2026-07-05T06:17:13.896978Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:17:13.896978Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Coding Social Science Datasets with Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Christopher Michael Rytting, David Wingate, Ethan Busby, Joshua Gubler, Lisa Argyle, Nancy Fulda, Taylor Sorensen","submitted_at":"2023-06-03T19:11:34Z","abstract_excerpt":"Researchers often rely on humans to code (label, annotate, etc.) large sets of texts. This kind of human coding forms an important part of social science research, yet the coding process is both resource intensive and highly variable from application to application. In some cases, efforts to automate this process have achieved human-level accuracies, but to achieve this, these attempts frequently rely on thousands of hand-labeled training examples, which makes them inapplicable to small-scale research studies and costly for large ones. Recent advances in a specific kind of artificial intellige"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.02177","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.02177/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.02177","created_at":"2026-07-05T06:17:13.897033+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.02177v1","created_at":"2026-07-05T06:17:13.897033+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.02177","created_at":"2026-07-05T06:17:13.897033+00:00"},{"alias_kind":"pith_short_12","alias_value":"33PWX6OEEOMV","created_at":"2026-07-05T06:17:13.897033+00:00"},{"alias_kind":"pith_short_16","alias_value":"33PWX6OEEOMVLL4W","created_at":"2026-07-05T06:17:13.897033+00:00"},{"alias_kind":"pith_short_8","alias_value":"33PWX6OE","created_at":"2026-07-05T06:17:13.897033+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.21786","citing_title":"From Codebooks to VLMs: Evaluating Automated Visual Discourse Analysis for Climate Change on Social Media","ref_index":145,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/33PWX6OEEOMVLL4WHOGPINU6DK","json":"https://pith.science/pith/33PWX6OEEOMVLL4WHOGPINU6DK.json","graph_json":"https://pith.science/api/pith-number/33PWX6OEEOMVLL4WHOGPINU6DK/graph.json","events_json":"https://pith.science/api/pith-number/33PWX6OEEOMVLL4WHOGPINU6DK/events.json","paper":"https://pith.science/paper/33PWX6OE"},"agent_actions":{"view_html":"https://pith.science/pith/33PWX6OEEOMVLL4WHOGPINU6DK","download_json":"https://pith.science/pith/33PWX6OEEOMVLL4WHOGPINU6DK.json","view_paper":"https://pith.science/paper/33PWX6OE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.02177&json=true","fetch_graph":"https://pith.science/api/pith-number/33PWX6OEEOMVLL4WHOGPINU6DK/graph.json","fetch_events":"https://pith.science/api/pith-number/33PWX6OEEOMVLL4WHOGPINU6DK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/33PWX6OEEOMVLL4WHOGPINU6DK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/33PWX6OEEOMVLL4WHOGPINU6DK/action/storage_attestation","attest_author":"https://pith.science/pith/33PWX6OEEOMVLL4WHOGPINU6DK/action/author_attestation","sign_citation":"https://pith.science/pith/33PWX6OEEOMVLL4WHOGPINU6DK/action/citation_signature","submit_replication":"https://pith.science/pith/33PWX6OEEOMVLL4WHOGPINU6DK/action/replication_record"}},"created_at":"2026-07-05T06:17:13.897033+00:00","updated_at":"2026-07-05T06:17:13.897033+00:00"}