{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:OT47Y2WEXT4JBWGYLNZBVDIH2P","short_pith_number":"pith:OT47Y2WE","schema_version":"1.0","canonical_sha256":"74f9fc6ac4bcf890d8d85b721a8d07d3f64a56b1ae4fe4a7274884d28df4ee18","source":{"kind":"arxiv","id":"2201.05955","version":5},"attestation_state":"computed","paper":{"title":"WANLI: Worker and AI Collaboration for Natural Language Inference Dataset Creation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alisa Liu, Noah A. Smith, Swabha Swayamdipta, Yejin Choi","submitted_at":"2022-01-16T03:13:49Z","abstract_excerpt":"A recurring challenge of crowdsourcing NLP datasets at scale is that human writers often rely on repetitive patterns when crafting examples, leading to a lack of linguistic diversity. We introduce a novel approach for dataset creation based on worker and AI collaboration, which brings together the generative strength of language models and the evaluative strength of humans. Starting with an existing dataset, MultiNLI for natural language inference (NLI), our approach uses dataset cartography to automatically identify examples that demonstrate challenging reasoning patterns, and instructs GPT-3"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2201.05955","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-01-16T03:13:49Z","cross_cats_sorted":[],"title_canon_sha256":"900e8e0610a3d4dcc601228be1cf58c798899e8297b96be828267e232eb4e237","abstract_canon_sha256":"86514227f985170897c538adec0c3f62866dfcf2e611e5c9b22a3f031f15e44d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:15:58.521971Z","signature_b64":"XuOqjFk8Nm5TKp0dI7US4Kd/mmJ8sxOZxZcEEOP01ThpiQ9iOpZqfCtyCLLj+tFw5WBC5z7wVU+aF0ND3CChDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"74f9fc6ac4bcf890d8d85b721a8d07d3f64a56b1ae4fe4a7274884d28df4ee18","last_reissued_at":"2026-07-05T05:15:58.521418Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:15:58.521418Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"WANLI: Worker and AI Collaboration for Natural Language Inference Dataset Creation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alisa Liu, Noah A. Smith, Swabha Swayamdipta, Yejin Choi","submitted_at":"2022-01-16T03:13:49Z","abstract_excerpt":"A recurring challenge of crowdsourcing NLP datasets at scale is that human writers often rely on repetitive patterns when crafting examples, leading to a lack of linguistic diversity. We introduce a novel approach for dataset creation based on worker and AI collaboration, which brings together the generative strength of language models and the evaluative strength of humans. Starting with an existing dataset, MultiNLI for natural language inference (NLI), our approach uses dataset cartography to automatically identify examples that demonstrate challenging reasoning patterns, and instructs GPT-3"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2201.05955","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2201.05955/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2201.05955","created_at":"2026-07-05T05:15:58.521484+00:00"},{"alias_kind":"arxiv_version","alias_value":"2201.05955v5","created_at":"2026-07-05T05:15:58.521484+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2201.05955","created_at":"2026-07-05T05:15:58.521484+00:00"},{"alias_kind":"pith_short_12","alias_value":"OT47Y2WEXT4J","created_at":"2026-07-05T05:15:58.521484+00:00"},{"alias_kind":"pith_short_16","alias_value":"OT47Y2WEXT4JBWGY","created_at":"2026-07-05T05:15:58.521484+00:00"},{"alias_kind":"pith_short_8","alias_value":"OT47Y2WE","created_at":"2026-07-05T05:15:58.521484+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20369","citing_title":"CATCH-ME if you RAG: a dataset of Contextually Annotated multi-Turn Counterspeech against Hate and Misinformation Exchanges","ref_index":189,"is_internal_anchor":false},{"citing_arxiv_id":"2301.13688","citing_title":"The Flan Collection: Designing Data and Methods for Effective Instruction Tuning","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22435","citing_title":"Assisted Counterspeech Writing at the Crossroads of Hate Speech and Misinformation","ref_index":178,"is_internal_anchor":false},{"citing_arxiv_id":"2303.09014","citing_title":"ART: Automatic multi-step reasoning and tool-use for large language models","ref_index":42,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OT47Y2WEXT4JBWGYLNZBVDIH2P","json":"https://pith.science/pith/OT47Y2WEXT4JBWGYLNZBVDIH2P.json","graph_json":"https://pith.science/api/pith-number/OT47Y2WEXT4JBWGYLNZBVDIH2P/graph.json","events_json":"https://pith.science/api/pith-number/OT47Y2WEXT4JBWGYLNZBVDIH2P/events.json","paper":"https://pith.science/paper/OT47Y2WE"},"agent_actions":{"view_html":"https://pith.science/pith/OT47Y2WEXT4JBWGYLNZBVDIH2P","download_json":"https://pith.science/pith/OT47Y2WEXT4JBWGYLNZBVDIH2P.json","view_paper":"https://pith.science/paper/OT47Y2WE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2201.05955&json=true","fetch_graph":"https://pith.science/api/pith-number/OT47Y2WEXT4JBWGYLNZBVDIH2P/graph.json","fetch_events":"https://pith.science/api/pith-number/OT47Y2WEXT4JBWGYLNZBVDIH2P/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OT47Y2WEXT4JBWGYLNZBVDIH2P/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OT47Y2WEXT4JBWGYLNZBVDIH2P/action/storage_attestation","attest_author":"https://pith.science/pith/OT47Y2WEXT4JBWGYLNZBVDIH2P/action/author_attestation","sign_citation":"https://pith.science/pith/OT47Y2WEXT4JBWGYLNZBVDIH2P/action/citation_signature","submit_replication":"https://pith.science/pith/OT47Y2WEXT4JBWGYLNZBVDIH2P/action/replication_record"}},"created_at":"2026-07-05T05:15:58.521484+00:00","updated_at":"2026-07-05T05:15:58.521484+00:00"}