{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:IM6CHVOUEO6F6HL5FLCBTQN75H","short_pith_number":"pith:IM6CHVOU","schema_version":"1.0","canonical_sha256":"433c23d5d423bc5f1d7d2ac419c1bfe9ffea2f57f893f8942ada97d327a46e28","source":{"kind":"arxiv","id":"2105.03075","version":5},"attestation_state":"computed","paper":{"title":"A Survey of Data Augmentation Approaches for NLP","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Eduard Hovy, Jason Wei, Sarath Chandar, Soroush Vosoughi, Steven Y. Feng, Teruko Mitamura, Varun Gangal","submitted_at":"2021-05-07T06:03:45Z","abstract_excerpt":"Data augmentation has recently seen increased interest in NLP due to more work in low-resource domains, new tasks, and the popularity of large-scale neural networks that require large amounts of training data. Despite this recent upsurge, this area is still relatively underexplored, perhaps due to the challenges posed by the discrete nature of language data. In this paper, we present a comprehensive and unifying survey of data augmentation for NLP by summarizing the literature in a structured manner. We first introduce and motivate data augmentation for NLP, and then discuss major methodologic"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2105.03075","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-05-07T06:03:45Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"7986134240f436ad1e82f29cc3500cefca39b390ae62997774657e5d5741f49c","abstract_canon_sha256":"56a0ff982079e13b3149578dcc662d488295ada2a389c512acb6dcdc78d3e73d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:36:54.250227Z","signature_b64":"oC0zGz7lu3hDNdeQhzsA45lN4D49RSXmUFpAY6AfOC/rg21tn3BBPOh2Mon9Bt7tTvIH3PBcUVUTnwyKtKBKDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"433c23d5d423bc5f1d7d2ac419c1bfe9ffea2f57f893f8942ada97d327a46e28","last_reissued_at":"2026-07-05T03:36:54.249713Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:36:54.249713Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Survey of Data Augmentation Approaches for NLP","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Eduard Hovy, Jason Wei, Sarath Chandar, Soroush Vosoughi, Steven Y. Feng, Teruko Mitamura, Varun Gangal","submitted_at":"2021-05-07T06:03:45Z","abstract_excerpt":"Data augmentation has recently seen increased interest in NLP due to more work in low-resource domains, new tasks, and the popularity of large-scale neural networks that require large amounts of training data. Despite this recent upsurge, this area is still relatively underexplored, perhaps due to the challenges posed by the discrete nature of language data. In this paper, we present a comprehensive and unifying survey of data augmentation for NLP by summarizing the literature in a structured manner. We first introduce and motivate data augmentation for NLP, and then discuss major methodologic"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2105.03075","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2105.03075/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2105.03075","created_at":"2026-07-05T03:36:54.249775+00:00"},{"alias_kind":"arxiv_version","alias_value":"2105.03075v5","created_at":"2026-07-05T03:36:54.249775+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2105.03075","created_at":"2026-07-05T03:36:54.249775+00:00"},{"alias_kind":"pith_short_12","alias_value":"IM6CHVOUEO6F","created_at":"2026-07-05T03:36:54.249775+00:00"},{"alias_kind":"pith_short_16","alias_value":"IM6CHVOUEO6F6HL5","created_at":"2026-07-05T03:36:54.249775+00:00"},{"alias_kind":"pith_short_8","alias_value":"IM6CHVOU","created_at":"2026-07-05T03:36:54.249775+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2411.18084","citing_title":"From Exploration to Revelation: Detecting Dark Patterns in Mobile Apps","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2411.18104","citing_title":"Training and Evaluating Language Models with Template-based Data Generation","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10290","citing_title":"Characterizing the Generalization Error of Random Feature Regression with Arbitrary Data-Augmentation","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IM6CHVOUEO6F6HL5FLCBTQN75H","json":"https://pith.science/pith/IM6CHVOUEO6F6HL5FLCBTQN75H.json","graph_json":"https://pith.science/api/pith-number/IM6CHVOUEO6F6HL5FLCBTQN75H/graph.json","events_json":"https://pith.science/api/pith-number/IM6CHVOUEO6F6HL5FLCBTQN75H/events.json","paper":"https://pith.science/paper/IM6CHVOU"},"agent_actions":{"view_html":"https://pith.science/pith/IM6CHVOUEO6F6HL5FLCBTQN75H","download_json":"https://pith.science/pith/IM6CHVOUEO6F6HL5FLCBTQN75H.json","view_paper":"https://pith.science/paper/IM6CHVOU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2105.03075&json=true","fetch_graph":"https://pith.science/api/pith-number/IM6CHVOUEO6F6HL5FLCBTQN75H/graph.json","fetch_events":"https://pith.science/api/pith-number/IM6CHVOUEO6F6HL5FLCBTQN75H/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IM6CHVOUEO6F6HL5FLCBTQN75H/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IM6CHVOUEO6F6HL5FLCBTQN75H/action/storage_attestation","attest_author":"https://pith.science/pith/IM6CHVOUEO6F6HL5FLCBTQN75H/action/author_attestation","sign_citation":"https://pith.science/pith/IM6CHVOUEO6F6HL5FLCBTQN75H/action/citation_signature","submit_replication":"https://pith.science/pith/IM6CHVOUEO6F6HL5FLCBTQN75H/action/replication_record"}},"created_at":"2026-07-05T03:36:54.249775+00:00","updated_at":"2026-07-05T03:36:54.249775+00:00"}