{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:5QG7PZ27OSR54MNTTIDRZ5KU2G","short_pith_number":"pith:5QG7PZ27","schema_version":"1.0","canonical_sha256":"ec0df7e75f74a3de31b39a071cf554d19ba563f63e2022ec032a982e43d42bc9","source":{"kind":"arxiv","id":"2205.11306","version":1},"attestation_state":"computed","paper":{"title":"Sample Efficient Approaches for Idiomaticity Detection","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Aline Villavicencio, Carolina Scarton, Dylan Phelps, Edward Gow-Smith, Harish Tayyar Madabushi, Xuan-Rui Fan","submitted_at":"2022-05-23T13:46:35Z","abstract_excerpt":"Deep neural models, in particular Transformer-based pre-trained language models, require a significant amount of data to train. This need for data tends to lead to problems when dealing with idiomatic multiword expressions (MWEs), which are inherently less frequent in natural text. As such, this work explores sample efficient methods of idiomaticity detection. In particular we study the impact of Pattern Exploit Training (PET), a few-shot method of classification, and BERTRAM, an efficient method of creating contextual embeddings, on the task of idiomaticity detection. In addition, to further "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2205.11306","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-05-23T13:46:35Z","cross_cats_sorted":[],"title_canon_sha256":"75dd5b26749710747e298bd95658677798aa3783ae0c3f0b5120e741d0bab27f","abstract_canon_sha256":"08343c0044968cf1ef871d8eaaa707f26133bacac5a2be54fcc6de9c888840e5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:25:29.337046Z","signature_b64":"bj2ju1Rvl0prU0yDMjYsu9vKfraAD5tF2/xABGpBQ1xY5NBedjmo5xCP46VO6l295OzkgY9lssEJHOl+JCHMAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ec0df7e75f74a3de31b39a071cf554d19ba563f63e2022ec032a982e43d42bc9","last_reissued_at":"2026-07-05T04:25:29.336637Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:25:29.336637Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Sample Efficient Approaches for Idiomaticity Detection","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Aline Villavicencio, Carolina Scarton, Dylan Phelps, Edward Gow-Smith, Harish Tayyar Madabushi, Xuan-Rui Fan","submitted_at":"2022-05-23T13:46:35Z","abstract_excerpt":"Deep neural models, in particular Transformer-based pre-trained language models, require a significant amount of data to train. This need for data tends to lead to problems when dealing with idiomatic multiword expressions (MWEs), which are inherently less frequent in natural text. As such, this work explores sample efficient methods of idiomaticity detection. In particular we study the impact of Pattern Exploit Training (PET), a few-shot method of classification, and BERTRAM, an efficient method of creating contextual embeddings, on the task of idiomaticity detection. In addition, to further "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2205.11306","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2205.11306/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2205.11306","created_at":"2026-07-05T04:25:29.336697+00:00"},{"alias_kind":"arxiv_version","alias_value":"2205.11306v1","created_at":"2026-07-05T04:25:29.336697+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2205.11306","created_at":"2026-07-05T04:25:29.336697+00:00"},{"alias_kind":"pith_short_12","alias_value":"5QG7PZ27OSR5","created_at":"2026-07-05T04:25:29.336697+00:00"},{"alias_kind":"pith_short_16","alias_value":"5QG7PZ27OSR54MNT","created_at":"2026-07-05T04:25:29.336697+00:00"},{"alias_kind":"pith_short_8","alias_value":"5QG7PZ27","created_at":"2026-07-05T04:25:29.336697+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5QG7PZ27OSR54MNTTIDRZ5KU2G","json":"https://pith.science/pith/5QG7PZ27OSR54MNTTIDRZ5KU2G.json","graph_json":"https://pith.science/api/pith-number/5QG7PZ27OSR54MNTTIDRZ5KU2G/graph.json","events_json":"https://pith.science/api/pith-number/5QG7PZ27OSR54MNTTIDRZ5KU2G/events.json","paper":"https://pith.science/paper/5QG7PZ27"},"agent_actions":{"view_html":"https://pith.science/pith/5QG7PZ27OSR54MNTTIDRZ5KU2G","download_json":"https://pith.science/pith/5QG7PZ27OSR54MNTTIDRZ5KU2G.json","view_paper":"https://pith.science/paper/5QG7PZ27","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2205.11306&json=true","fetch_graph":"https://pith.science/api/pith-number/5QG7PZ27OSR54MNTTIDRZ5KU2G/graph.json","fetch_events":"https://pith.science/api/pith-number/5QG7PZ27OSR54MNTTIDRZ5KU2G/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5QG7PZ27OSR54MNTTIDRZ5KU2G/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5QG7PZ27OSR54MNTTIDRZ5KU2G/action/storage_attestation","attest_author":"https://pith.science/pith/5QG7PZ27OSR54MNTTIDRZ5KU2G/action/author_attestation","sign_citation":"https://pith.science/pith/5QG7PZ27OSR54MNTTIDRZ5KU2G/action/citation_signature","submit_replication":"https://pith.science/pith/5QG7PZ27OSR54MNTTIDRZ5KU2G/action/replication_record"}},"created_at":"2026-07-05T04:25:29.336697+00:00","updated_at":"2026-07-05T04:25:29.336697+00:00"}