{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2018:W2TAQSIRZ62UWOLBKF3ZRXOETU","short_pith_number":"pith:W2TAQSIR","schema_version":"1.0","canonical_sha256":"b6a6084911cfb54b3961517798ddc49d16aad89ccb5fdb8febbb225f11ec72fa","source":{"kind":"arxiv","id":"1803.02324","version":2},"attestation_state":"computed","paper":{"title":"Annotation Artifacts in Natural Language Inference Data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Noah A. Smith, Omer Levy, Roy Schwartz, Samuel R. Bowman, Suchin Gururangan, Swabha Swayamdipta","submitted_at":"2018-03-06T18:23:08Z","abstract_excerpt":"Large-scale datasets for natural language inference are created by presenting crowd workers with a sentence (premise), and asking them to generate three new sentences (hypotheses) that it entails, contradicts, or is logically neutral with respect to. We show that, in a significant portion of such data, this protocol leaves clues that make it possible to identify the label by looking only at the hypothesis, without observing the premise. Specifically, we show that a simple text categorization model can correctly classify the hypothesis alone in about 67% of SNLI (Bowman et. al, 2015) and 53% of"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1803.02324","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2018-03-06T18:23:08Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ed8311614f8ee3e1fbae411f98f87ce5d4048f87da19acddf3e284d14c93aab9","abstract_canon_sha256":"8a194e183a0a189383fed6738caf88c7f3155888399061cb5645a1258a934348"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T00:18:25.175274Z","signature_b64":"8lcigD6FpMEYjdLhVXJ9x4hT88VT+Zyl7RmdYjuONgr7NCxpLphfeyHnXeThzfBiKsDsCzkFqMwP87qCWLgZCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b6a6084911cfb54b3961517798ddc49d16aad89ccb5fdb8febbb225f11ec72fa","last_reissued_at":"2026-05-18T00:18:25.174708Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T00:18:25.174708Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Annotation Artifacts in Natural Language Inference Data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Noah A. Smith, Omer Levy, Roy Schwartz, Samuel R. Bowman, Suchin Gururangan, Swabha Swayamdipta","submitted_at":"2018-03-06T18:23:08Z","abstract_excerpt":"Large-scale datasets for natural language inference are created by presenting crowd workers with a sentence (premise), and asking them to generate three new sentences (hypotheses) that it entails, contradicts, or is logically neutral with respect to. We show that, in a significant portion of such data, this protocol leaves clues that make it possible to identify the label by looking only at the hypothesis, without observing the premise. Specifically, we show that a simple text categorization model can correctly classify the hypothesis alone in about 67% of SNLI (Bowman et. al, 2015) and 53% of"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1803.02324","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1803.02324","created_at":"2026-05-18T00:18:25.174791+00:00"},{"alias_kind":"arxiv_version","alias_value":"1803.02324v2","created_at":"2026-05-18T00:18:25.174791+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1803.02324","created_at":"2026-05-18T00:18:25.174791+00:00"},{"alias_kind":"pith_short_12","alias_value":"W2TAQSIRZ62U","created_at":"2026-05-18T12:32:59.047623+00:00"},{"alias_kind":"pith_short_16","alias_value":"W2TAQSIRZ62UWOLB","created_at":"2026-05-18T12:32:59.047623+00:00"},{"alias_kind":"pith_short_8","alias_value":"W2TAQSIR","created_at":"2026-05-18T12:32:59.047623+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":4,"sample":[{"citing_arxiv_id":"2605.30462","citing_title":"idSCD: Identifying Training Datasets through Semantic Correlation Descriptors","ref_index":16,"is_internal_anchor":true},{"citing_arxiv_id":"1906.09635","citing_title":"Investigating Biases in Textual Entailment Datasets","ref_index":7,"is_internal_anchor":true},{"citing_arxiv_id":"2303.09014","citing_title":"ART: Automatic multi-step reasoning and tool-use for large language models","ref_index":104,"is_internal_anchor":true},{"citing_arxiv_id":"1909.01066","citing_title":"Language Models as Knowledge Bases?","ref_index":213,"is_internal_anchor":true},{"citing_arxiv_id":"2305.14314","citing_title":"QLoRA: Efficient Finetuning of Quantized LLMs","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2005.14165","citing_title":"Language Models are Few-Shot Learners","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/W2TAQSIRZ62UWOLBKF3ZRXOETU","json":"https://pith.science/pith/W2TAQSIRZ62UWOLBKF3ZRXOETU.json","graph_json":"https://pith.science/api/pith-number/W2TAQSIRZ62UWOLBKF3ZRXOETU/graph.json","events_json":"https://pith.science/api/pith-number/W2TAQSIRZ62UWOLBKF3ZRXOETU/events.json","paper":"https://pith.science/paper/W2TAQSIR"},"agent_actions":{"view_html":"https://pith.science/pith/W2TAQSIRZ62UWOLBKF3ZRXOETU","download_json":"https://pith.science/pith/W2TAQSIRZ62UWOLBKF3ZRXOETU.json","view_paper":"https://pith.science/paper/W2TAQSIR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1803.02324&json=true","fetch_graph":"https://pith.science/api/pith-number/W2TAQSIRZ62UWOLBKF3ZRXOETU/graph.json","fetch_events":"https://pith.science/api/pith-number/W2TAQSIRZ62UWOLBKF3ZRXOETU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/W2TAQSIRZ62UWOLBKF3ZRXOETU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/W2TAQSIRZ62UWOLBKF3ZRXOETU/action/storage_attestation","attest_author":"https://pith.science/pith/W2TAQSIRZ62UWOLBKF3ZRXOETU/action/author_attestation","sign_citation":"https://pith.science/pith/W2TAQSIRZ62UWOLBKF3ZRXOETU/action/citation_signature","submit_replication":"https://pith.science/pith/W2TAQSIRZ62UWOLBKF3ZRXOETU/action/replication_record"}},"created_at":"2026-05-18T00:18:25.174791+00:00","updated_at":"2026-05-18T00:18:25.174791+00:00"}