{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:J6J2KC2DBXNMZ2XF5OV7TC2M33","short_pith_number":"pith:J6J2KC2D","schema_version":"1.0","canonical_sha256":"4f93a50b430ddacceae5ebabf98b4cdecbf56ff954a33cd446bb46d9ebcf4a49","source":{"kind":"arxiv","id":"2102.01335","version":1},"attestation_state":"computed","paper":{"title":"Neural Data Augmentation via Example Extrapolation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Hyung Won Chung, Kelvin Guu, Kenton Lee, Luheng He, Tim Dozat","submitted_at":"2021-02-02T06:20:19Z","abstract_excerpt":"In many applications of machine learning, certain categories of examples may be underrepresented in the training data, causing systems to underperform on such \"few-shot\" cases at test time. A common remedy is to perform data augmentation, such as by duplicating underrepresented examples, or heuristically synthesizing new examples. But these remedies often fail to cover the full diversity and complexity of real examples.\n  We propose a data augmentation approach that performs neural Example Extrapolation (Ex2). Given a handful of exemplars sampled from some distribution, Ex2 synthesizes new exa"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2102.01335","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-02-02T06:20:19Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"d0d600173f6fc583f86067e742cfead59ab17a87e1494bc6675bf468ece69d15","abstract_canon_sha256":"3e6dc41a4672668de9ed3912e1bd713c35dac3ab78ff60bb0b8ffdb49074688b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:11:29.673527Z","signature_b64":"Wqy/s7yz0Hdq/1KMVMMZUTZCl693GdCXWuuTD7Nwspy6zA5D5EuqYVybuBxh7KUGnKhb5SE+6X9RZVt/flIWDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4f93a50b430ddacceae5ebabf98b4cdecbf56ff954a33cd446bb46d9ebcf4a49","last_reissued_at":"2026-07-05T02:11:29.673076Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:11:29.673076Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Neural Data Augmentation via Example Extrapolation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Hyung Won Chung, Kelvin Guu, Kenton Lee, Luheng He, Tim Dozat","submitted_at":"2021-02-02T06:20:19Z","abstract_excerpt":"In many applications of machine learning, certain categories of examples may be underrepresented in the training data, causing systems to underperform on such \"few-shot\" cases at test time. A common remedy is to perform data augmentation, such as by duplicating underrepresented examples, or heuristically synthesizing new examples. But these remedies often fail to cover the full diversity and complexity of real examples.\n  We propose a data augmentation approach that performs neural Example Extrapolation (Ex2). Given a handful of exemplars sampled from some distribution, Ex2 synthesizes new exa"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2102.01335","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2102.01335/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2102.01335","created_at":"2026-07-05T02:11:29.673133+00:00"},{"alias_kind":"arxiv_version","alias_value":"2102.01335v1","created_at":"2026-07-05T02:11:29.673133+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2102.01335","created_at":"2026-07-05T02:11:29.673133+00:00"},{"alias_kind":"pith_short_12","alias_value":"J6J2KC2DBXNM","created_at":"2026-07-05T02:11:29.673133+00:00"},{"alias_kind":"pith_short_16","alias_value":"J6J2KC2DBXNMZ2XF","created_at":"2026-07-05T02:11:29.673133+00:00"},{"alias_kind":"pith_short_8","alias_value":"J6J2KC2D","created_at":"2026-07-05T02:11:29.673133+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2104.08773","citing_title":"Cross-Task Generalization via Natural Language Crowdsourcing Instructions","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2406.10162","citing_title":"Sycophancy to Subterfuge: Investigating Reward-Tampering in Large Language Models","ref_index":216,"is_internal_anchor":false},{"citing_arxiv_id":"2303.17760","citing_title":"CAMEL: Communicative Agents for \"Mind\" Exploration of Large Language Model Society","ref_index":63,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/J6J2KC2DBXNMZ2XF5OV7TC2M33","json":"https://pith.science/pith/J6J2KC2DBXNMZ2XF5OV7TC2M33.json","graph_json":"https://pith.science/api/pith-number/J6J2KC2DBXNMZ2XF5OV7TC2M33/graph.json","events_json":"https://pith.science/api/pith-number/J6J2KC2DBXNMZ2XF5OV7TC2M33/events.json","paper":"https://pith.science/paper/J6J2KC2D"},"agent_actions":{"view_html":"https://pith.science/pith/J6J2KC2DBXNMZ2XF5OV7TC2M33","download_json":"https://pith.science/pith/J6J2KC2DBXNMZ2XF5OV7TC2M33.json","view_paper":"https://pith.science/paper/J6J2KC2D","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2102.01335&json=true","fetch_graph":"https://pith.science/api/pith-number/J6J2KC2DBXNMZ2XF5OV7TC2M33/graph.json","fetch_events":"https://pith.science/api/pith-number/J6J2KC2DBXNMZ2XF5OV7TC2M33/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/J6J2KC2DBXNMZ2XF5OV7TC2M33/action/timestamp_anchor","attest_storage":"https://pith.science/pith/J6J2KC2DBXNMZ2XF5OV7TC2M33/action/storage_attestation","attest_author":"https://pith.science/pith/J6J2KC2DBXNMZ2XF5OV7TC2M33/action/author_attestation","sign_citation":"https://pith.science/pith/J6J2KC2DBXNMZ2XF5OV7TC2M33/action/citation_signature","submit_replication":"https://pith.science/pith/J6J2KC2DBXNMZ2XF5OV7TC2M33/action/replication_record"}},"created_at":"2026-07-05T02:11:29.673133+00:00","updated_at":"2026-07-05T02:11:29.673133+00:00"}