{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:D2XHR3ZOZWBMMDXWUXEAV4EFUO","short_pith_number":"pith:D2XHR3ZO","schema_version":"1.0","canonical_sha256":"1eae78ef2ecd82c60ef6a5c80af085a3bcbf4d901cbe0cb486a10762a485c860","source":{"kind":"arxiv","id":"2010.02114","version":4},"attestation_state":"computed","paper":{"title":"Explaining The Efficacy of Counterfactually Augmented Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","stat.ML"],"primary_cat":"cs.CL","authors_text":"Amrith Setlur, Divyansh Kaushik, Eduard Hovy, Zachary C. Lipton","submitted_at":"2020-10-05T15:57:07Z","abstract_excerpt":"In attempts to produce ML models less reliant on spurious patterns in NLP datasets, researchers have recently proposed curating counterfactually augmented data (CAD) via a human-in-the-loop process in which given some documents and their (initial) labels, humans must revise the text to make a counterfactual label applicable. Importantly, edits that are not necessary to flip the applicable label are prohibited. Models trained on the augmented data appear, empirically, to rely less on semantically irrelevant words and to generalize better out of domain. While this work draws loosely on causal th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2010.02114","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-10-05T15:57:07Z","cross_cats_sorted":["cs.AI","cs.LG","stat.ML"],"title_canon_sha256":"72d6b6187e23c8181bf338567ddd533e1989590dbf8e360c350fe473ea704fa8","abstract_canon_sha256":"852acb030a401cd3446f07f949dc8376b04824f4243721475c73d187ac5a0056"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:25:50.310414Z","signature_b64":"jbFuSS2lhHNmF+7w4D0WVmwpaMz8/JrMnv21iUu/AC2+xW0DTL/78btj3RWXOzWC08u/xidMUj5qdZ6IZkzRDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1eae78ef2ecd82c60ef6a5c80af085a3bcbf4d901cbe0cb486a10762a485c860","last_reissued_at":"2026-07-05T02:25:50.309929Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:25:50.309929Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Explaining The Efficacy of Counterfactually Augmented Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","stat.ML"],"primary_cat":"cs.CL","authors_text":"Amrith Setlur, Divyansh Kaushik, Eduard Hovy, Zachary C. Lipton","submitted_at":"2020-10-05T15:57:07Z","abstract_excerpt":"In attempts to produce ML models less reliant on spurious patterns in NLP datasets, researchers have recently proposed curating counterfactually augmented data (CAD) via a human-in-the-loop process in which given some documents and their (initial) labels, humans must revise the text to make a counterfactual label applicable. Importantly, edits that are not necessary to flip the applicable label are prohibited. Models trained on the augmented data appear, empirically, to rely less on semantically irrelevant words and to generalize better out of domain. While this work draws loosely on causal th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2010.02114","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2010.02114/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2010.02114","created_at":"2026-07-05T02:25:50.309987+00:00"},{"alias_kind":"arxiv_version","alias_value":"2010.02114v4","created_at":"2026-07-05T02:25:50.309987+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2010.02114","created_at":"2026-07-05T02:25:50.309987+00:00"},{"alias_kind":"pith_short_12","alias_value":"D2XHR3ZOZWBM","created_at":"2026-07-05T02:25:50.309987+00:00"},{"alias_kind":"pith_short_16","alias_value":"D2XHR3ZOZWBMMDXW","created_at":"2026-07-05T02:25:50.309987+00:00"},{"alias_kind":"pith_short_8","alias_value":"D2XHR3ZO","created_at":"2026-07-05T02:25:50.309987+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.09716","citing_title":"Medical Model Synthesis Architectures: A Case Study","ref_index":184,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/D2XHR3ZOZWBMMDXWUXEAV4EFUO","json":"https://pith.science/pith/D2XHR3ZOZWBMMDXWUXEAV4EFUO.json","graph_json":"https://pith.science/api/pith-number/D2XHR3ZOZWBMMDXWUXEAV4EFUO/graph.json","events_json":"https://pith.science/api/pith-number/D2XHR3ZOZWBMMDXWUXEAV4EFUO/events.json","paper":"https://pith.science/paper/D2XHR3ZO"},"agent_actions":{"view_html":"https://pith.science/pith/D2XHR3ZOZWBMMDXWUXEAV4EFUO","download_json":"https://pith.science/pith/D2XHR3ZOZWBMMDXWUXEAV4EFUO.json","view_paper":"https://pith.science/paper/D2XHR3ZO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2010.02114&json=true","fetch_graph":"https://pith.science/api/pith-number/D2XHR3ZOZWBMMDXWUXEAV4EFUO/graph.json","fetch_events":"https://pith.science/api/pith-number/D2XHR3ZOZWBMMDXWUXEAV4EFUO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/D2XHR3ZOZWBMMDXWUXEAV4EFUO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/D2XHR3ZOZWBMMDXWUXEAV4EFUO/action/storage_attestation","attest_author":"https://pith.science/pith/D2XHR3ZOZWBMMDXWUXEAV4EFUO/action/author_attestation","sign_citation":"https://pith.science/pith/D2XHR3ZOZWBMMDXWUXEAV4EFUO/action/citation_signature","submit_replication":"https://pith.science/pith/D2XHR3ZOZWBMMDXWUXEAV4EFUO/action/replication_record"}},"created_at":"2026-07-05T02:25:50.309987+00:00","updated_at":"2026-07-05T02:25:50.309987+00:00"}