{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:POESHTOGLRM7KWDLF4CXE5X7NM","short_pith_number":"pith:POESHTOG","schema_version":"1.0","canonical_sha256":"7b8923cdc65c59f5586b2f057276ff6b14845b503fde56529de32a1df271e763","source":{"kind":"arxiv","id":"2105.02692","version":3},"attestation_state":"computed","paper":{"title":"Learning to Perturb Word Embeddings for Out-of-distribution QA","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Juho Lee, Minki Kang, Seanie Lee, Sung Ju Hwang","submitted_at":"2021-05-06T14:12:26Z","abstract_excerpt":"QA models based on pretrained language mod-els have achieved remarkable performance on various benchmark datasets.However, QA models do not generalize well to unseen data that falls outside the training distribution, due to distributional shifts.Data augmentation (DA) techniques which drop/replace words have shown to be effective in regularizing the model from overfitting to the training data.Yet, they may adversely affect the QA tasks since they incur semantic changes that may lead to wrong answers for the QA task. To tackle this problem, we propose a simple yet effective DA method based on a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2105.02692","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2021-05-06T14:12:26Z","cross_cats_sorted":[],"title_canon_sha256":"2e60880b1c9096c8a704b8d9b21ee59b159129b84d3408f4708942d6201b719c","abstract_canon_sha256":"ddbfb1c170ccdd2833e49b92d18af371e016bbca1db403017d5983665e7fe2af"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:52:02.623554Z","signature_b64":"b5Ot3L95IRQVtok8A+tgp4xpzJ2OpbDUg3U8BwgK2trqzGxhM+qshg4YMEjvay+Zl4ECV5fkaKcvlzN3avjLCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7b8923cdc65c59f5586b2f057276ff6b14845b503fde56529de32a1df271e763","last_reissued_at":"2026-07-05T02:52:02.623023Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:52:02.623023Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning to Perturb Word Embeddings for Out-of-distribution QA","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Juho Lee, Minki Kang, Seanie Lee, Sung Ju Hwang","submitted_at":"2021-05-06T14:12:26Z","abstract_excerpt":"QA models based on pretrained language mod-els have achieved remarkable performance on various benchmark datasets.However, QA models do not generalize well to unseen data that falls outside the training distribution, due to distributional shifts.Data augmentation (DA) techniques which drop/replace words have shown to be effective in regularizing the model from overfitting to the training data.Yet, they may adversely affect the QA tasks since they incur semantic changes that may lead to wrong answers for the QA task. To tackle this problem, we propose a simple yet effective DA method based on a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2105.02692","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2105.02692/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2105.02692","created_at":"2026-07-05T02:52:02.623083+00:00"},{"alias_kind":"arxiv_version","alias_value":"2105.02692v3","created_at":"2026-07-05T02:52:02.623083+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2105.02692","created_at":"2026-07-05T02:52:02.623083+00:00"},{"alias_kind":"pith_short_12","alias_value":"POESHTOGLRM7","created_at":"2026-07-05T02:52:02.623083+00:00"},{"alias_kind":"pith_short_16","alias_value":"POESHTOGLRM7KWDL","created_at":"2026-07-05T02:52:02.623083+00:00"},{"alias_kind":"pith_short_8","alias_value":"POESHTOG","created_at":"2026-07-05T02:52:02.623083+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.21848","citing_title":"FPAN: Mitigating Replication in Diffusion Models through the Fine-Grained Probabilistic Addition of Noise to Token Embeddings","ref_index":13,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/POESHTOGLRM7KWDLF4CXE5X7NM","json":"https://pith.science/pith/POESHTOGLRM7KWDLF4CXE5X7NM.json","graph_json":"https://pith.science/api/pith-number/POESHTOGLRM7KWDLF4CXE5X7NM/graph.json","events_json":"https://pith.science/api/pith-number/POESHTOGLRM7KWDLF4CXE5X7NM/events.json","paper":"https://pith.science/paper/POESHTOG"},"agent_actions":{"view_html":"https://pith.science/pith/POESHTOGLRM7KWDLF4CXE5X7NM","download_json":"https://pith.science/pith/POESHTOGLRM7KWDLF4CXE5X7NM.json","view_paper":"https://pith.science/paper/POESHTOG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2105.02692&json=true","fetch_graph":"https://pith.science/api/pith-number/POESHTOGLRM7KWDLF4CXE5X7NM/graph.json","fetch_events":"https://pith.science/api/pith-number/POESHTOGLRM7KWDLF4CXE5X7NM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/POESHTOGLRM7KWDLF4CXE5X7NM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/POESHTOGLRM7KWDLF4CXE5X7NM/action/storage_attestation","attest_author":"https://pith.science/pith/POESHTOGLRM7KWDLF4CXE5X7NM/action/author_attestation","sign_citation":"https://pith.science/pith/POESHTOGLRM7KWDLF4CXE5X7NM/action/citation_signature","submit_replication":"https://pith.science/pith/POESHTOGLRM7KWDLF4CXE5X7NM/action/replication_record"}},"created_at":"2026-07-05T02:52:02.623083+00:00","updated_at":"2026-07-05T02:52:02.623083+00:00"}