{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2018:7XZZNGIJFUIHH4FQZIFUJRP3FM","short_pith_number":"pith:7XZZNGIJ","schema_version":"1.0","canonical_sha256":"fdf39699092d1073f0b0ca0b44c5fb2b0d5bf91b8219b28e4b0b7641e2890d48","source":{"kind":"arxiv","id":"1812.04718","version":1},"attestation_state":"computed","paper":{"title":"Text Data Augmentation Made Simple By Leveraging NLP Cloud APIs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Claude Coulombe","submitted_at":"2018-12-05T03:59:41Z","abstract_excerpt":"In practice, it is common to find oneself with far too little text data to train a deep neural network. This \"Big Data Wall\" represents a challenge for minority language communities on the Internet, organizations, laboratories and companies that compete the GAFAM (Google, Amazon, Facebook, Apple, Microsoft). While most of the research effort in text data augmentation aims on the long-term goal of finding end-to-end learning solutions, which is equivalent to \"using neural networks to feed neural networks\", this engineering work focuses on the use of practical, robust, scalable and easy-to-imple"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1812.04718","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2018-12-05T03:59:41Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"19570e34f50424250ba84ed6a6b7d6a83f68f79e8e020276866d5566b691430f","abstract_canon_sha256":"a1f5a22162f3ef97961ca6a634a115317dce4ec584db1d200cc9bc6738532beb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-17T23:58:29.725522Z","signature_b64":"fFD/WnaKylkpf3Jl0sN4u2c0zhjOKQoDzYdXIVVGid9p5wwUAwthJQJFQ4Bsn85OVwZ3xepWf0ekfXeuJwQxDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fdf39699092d1073f0b0ca0b44c5fb2b0d5bf91b8219b28e4b0b7641e2890d48","last_reissued_at":"2026-05-17T23:58:29.724675Z","signature_status":"signed_v1","first_computed_at":"2026-05-17T23:58:29.724675Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Text Data Augmentation Made Simple By Leveraging NLP Cloud APIs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Claude Coulombe","submitted_at":"2018-12-05T03:59:41Z","abstract_excerpt":"In practice, it is common to find oneself with far too little text data to train a deep neural network. This \"Big Data Wall\" represents a challenge for minority language communities on the Internet, organizations, laboratories and companies that compete the GAFAM (Google, Amazon, Facebook, Apple, Microsoft). While most of the research effort in text data augmentation aims on the long-term goal of finding end-to-end learning solutions, which is equivalent to \"using neural networks to feed neural networks\", this engineering work focuses on the use of practical, robust, scalable and easy-to-imple"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1812.04718","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1812.04718","created_at":"2026-05-17T23:58:29.724811+00:00"},{"alias_kind":"arxiv_version","alias_value":"1812.04718v1","created_at":"2026-05-17T23:58:29.724811+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1812.04718","created_at":"2026-05-17T23:58:29.724811+00:00"},{"alias_kind":"pith_short_12","alias_value":"7XZZNGIJFUIH","created_at":"2026-05-18T12:32:13.499390+00:00"},{"alias_kind":"pith_short_16","alias_value":"7XZZNGIJFUIHH4FQ","created_at":"2026-05-18T12:32:13.499390+00:00"},{"alias_kind":"pith_short_8","alias_value":"7XZZNGIJ","created_at":"2026-05-18T12:32:13.499390+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.00891","citing_title":"MemeCMD: An Automatically Generated Chinese Multi-turn Dialogue Dataset with Contextually Retrieved Memes","ref_index":6,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7XZZNGIJFUIHH4FQZIFUJRP3FM","json":"https://pith.science/pith/7XZZNGIJFUIHH4FQZIFUJRP3FM.json","graph_json":"https://pith.science/api/pith-number/7XZZNGIJFUIHH4FQZIFUJRP3FM/graph.json","events_json":"https://pith.science/api/pith-number/7XZZNGIJFUIHH4FQZIFUJRP3FM/events.json","paper":"https://pith.science/paper/7XZZNGIJ"},"agent_actions":{"view_html":"https://pith.science/pith/7XZZNGIJFUIHH4FQZIFUJRP3FM","download_json":"https://pith.science/pith/7XZZNGIJFUIHH4FQZIFUJRP3FM.json","view_paper":"https://pith.science/paper/7XZZNGIJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1812.04718&json=true","fetch_graph":"https://pith.science/api/pith-number/7XZZNGIJFUIHH4FQZIFUJRP3FM/graph.json","fetch_events":"https://pith.science/api/pith-number/7XZZNGIJFUIHH4FQZIFUJRP3FM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7XZZNGIJFUIHH4FQZIFUJRP3FM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7XZZNGIJFUIHH4FQZIFUJRP3FM/action/storage_attestation","attest_author":"https://pith.science/pith/7XZZNGIJFUIHH4FQZIFUJRP3FM/action/author_attestation","sign_citation":"https://pith.science/pith/7XZZNGIJFUIHH4FQZIFUJRP3FM/action/citation_signature","submit_replication":"https://pith.science/pith/7XZZNGIJFUIHH4FQZIFUJRP3FM/action/replication_record"}},"created_at":"2026-05-17T23:58:29.724811+00:00","updated_at":"2026-05-17T23:58:29.724811+00:00"}