{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:GOOBZAYZTDTQB2VQ4VLMV2NW7M","short_pith_number":"pith:GOOBZAYZ","schema_version":"1.0","canonical_sha256":"339c1c831998e700eab0e556cae9b6fb339f6c25c8e19764b64e7bad5fe4554c","source":{"kind":"arxiv","id":"2101.05469","version":1},"attestation_state":"computed","paper":{"title":"Text Augmentation in a Multi-Task View","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chengyu Huang, Jason Wei, Shiqi Xu, Soroush Vosoughi","submitted_at":"2021-01-14T05:59:23Z","abstract_excerpt":"Traditional data augmentation aims to increase the coverage of the input distribution by generating augmented examples that strongly resemble original samples in an online fashion where augmented examples dominate training. In this paper, we propose an alternative perspective -- a multi-task view (MTV) of data augmentation -- in which the primary task trains on original examples and the auxiliary task trains on augmented examples. In MTV data augmentation, both original and augmented samples are weighted substantively during training, relaxing the constraint that augmented examples must resemb"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2101.05469","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2021-01-14T05:59:23Z","cross_cats_sorted":[],"title_canon_sha256":"d00fd59d60ca5cf3e2d8b8494e45ed58dcfb20fefb05d31509ae6cacf3ade1c9","abstract_canon_sha256":"9da3aa6f412c8905c19e13d1222c32d8e1b72fcb2333ed31df07f1384a05d6b8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:06:56.020488Z","signature_b64":"G1ldoJPstblB870mwQhS24ZdtkRafL3GOQMhNMzdwECAFDFjFf87vWFmKiAZp9jbz+SX4JEWUVLqR3xc85VaAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"339c1c831998e700eab0e556cae9b6fb339f6c25c8e19764b64e7bad5fe4554c","last_reissued_at":"2026-07-05T02:06:56.020108Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:06:56.020108Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Text Augmentation in a Multi-Task View","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chengyu Huang, Jason Wei, Shiqi Xu, Soroush Vosoughi","submitted_at":"2021-01-14T05:59:23Z","abstract_excerpt":"Traditional data augmentation aims to increase the coverage of the input distribution by generating augmented examples that strongly resemble original samples in an online fashion where augmented examples dominate training. In this paper, we propose an alternative perspective -- a multi-task view (MTV) of data augmentation -- in which the primary task trains on original examples and the auxiliary task trains on augmented examples. In MTV data augmentation, both original and augmented samples are weighted substantively during training, relaxing the constraint that augmented examples must resemb"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2101.05469","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2101.05469/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2101.05469","created_at":"2026-07-05T02:06:56.020166+00:00"},{"alias_kind":"arxiv_version","alias_value":"2101.05469v1","created_at":"2026-07-05T02:06:56.020166+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2101.05469","created_at":"2026-07-05T02:06:56.020166+00:00"},{"alias_kind":"pith_short_12","alias_value":"GOOBZAYZTDTQ","created_at":"2026-07-05T02:06:56.020166+00:00"},{"alias_kind":"pith_short_16","alias_value":"GOOBZAYZTDTQB2VQ","created_at":"2026-07-05T02:06:56.020166+00:00"},{"alias_kind":"pith_short_8","alias_value":"GOOBZAYZ","created_at":"2026-07-05T02:06:56.020166+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2504.15160","citing_title":"The Synthetic Imputation Approach: Generating Optimal Synthetic Texts For Underrepresented Categories In Supervised Classification Tasks","ref_index":8,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GOOBZAYZTDTQB2VQ4VLMV2NW7M","json":"https://pith.science/pith/GOOBZAYZTDTQB2VQ4VLMV2NW7M.json","graph_json":"https://pith.science/api/pith-number/GOOBZAYZTDTQB2VQ4VLMV2NW7M/graph.json","events_json":"https://pith.science/api/pith-number/GOOBZAYZTDTQB2VQ4VLMV2NW7M/events.json","paper":"https://pith.science/paper/GOOBZAYZ"},"agent_actions":{"view_html":"https://pith.science/pith/GOOBZAYZTDTQB2VQ4VLMV2NW7M","download_json":"https://pith.science/pith/GOOBZAYZTDTQB2VQ4VLMV2NW7M.json","view_paper":"https://pith.science/paper/GOOBZAYZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2101.05469&json=true","fetch_graph":"https://pith.science/api/pith-number/GOOBZAYZTDTQB2VQ4VLMV2NW7M/graph.json","fetch_events":"https://pith.science/api/pith-number/GOOBZAYZTDTQB2VQ4VLMV2NW7M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GOOBZAYZTDTQB2VQ4VLMV2NW7M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GOOBZAYZTDTQB2VQ4VLMV2NW7M/action/storage_attestation","attest_author":"https://pith.science/pith/GOOBZAYZTDTQB2VQ4VLMV2NW7M/action/author_attestation","sign_citation":"https://pith.science/pith/GOOBZAYZTDTQB2VQ4VLMV2NW7M/action/citation_signature","submit_replication":"https://pith.science/pith/GOOBZAYZTDTQB2VQ4VLMV2NW7M/action/replication_record"}},"created_at":"2026-07-05T02:06:56.020166+00:00","updated_at":"2026-07-05T02:06:56.020166+00:00"}