{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:YBEWXHEXQIRDZYXRDD6PVQ77GD","short_pith_number":"pith:YBEWXHEX","schema_version":"1.0","canonical_sha256":"c0496b9c9782223ce2f118fcfac3ff30c2f94ae755e84d38d903645373faf70c","source":{"kind":"arxiv","id":"2205.05050","version":1},"attestation_state":"computed","paper":{"title":"White-box Testing of NLP models with Mask Neuron Coverage","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Arshdeep Sekhon, Matthew B. Dwyer, Yangfeng Ji, Yanjun Qi","submitted_at":"2022-05-10T17:07:23Z","abstract_excerpt":"Recent literature has seen growing interest in using black-box strategies like CheckList for testing the behavior of NLP models. Research on white-box testing has developed a number of methods for evaluating how thoroughly the internal behavior of deep models is tested, but they are not applicable to NLP models. We propose a set of white-box testing methods that are customized for transformer-based NLP models. These include Mask Neuron Coverage (MNCOVER) that measures how thoroughly the attention layers in models are exercised during testing. We show that MNCOVER can refine testing suites gene"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2205.05050","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-05-10T17:07:23Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"41b290703d2871398a912cda816385c9322969a0fda0c6255e853d515ad8b51d","abstract_canon_sha256":"7a5a4e9c73f7f30c3b7eac21d6408a488a22ceb4deedd28167fe04d258498eeb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:22:13.344420Z","signature_b64":"BGFW6ecNZCGVYGDZWZyD+lC/MbeO7exYZzkIGf+ZFGcAJ6vtvDnFdxsgIFQJ9CIR3EBE8zhlQfQMfVoqnuZ7Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c0496b9c9782223ce2f118fcfac3ff30c2f94ae755e84d38d903645373faf70c","last_reissued_at":"2026-07-05T04:22:13.344039Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:22:13.344039Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"White-box Testing of NLP models with Mask Neuron Coverage","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Arshdeep Sekhon, Matthew B. Dwyer, Yangfeng Ji, Yanjun Qi","submitted_at":"2022-05-10T17:07:23Z","abstract_excerpt":"Recent literature has seen growing interest in using black-box strategies like CheckList for testing the behavior of NLP models. Research on white-box testing has developed a number of methods for evaluating how thoroughly the internal behavior of deep models is tested, but they are not applicable to NLP models. We propose a set of white-box testing methods that are customized for transformer-based NLP models. These include Mask Neuron Coverage (MNCOVER) that measures how thoroughly the attention layers in models are exercised during testing. We show that MNCOVER can refine testing suites gene"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2205.05050","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2205.05050/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2205.05050","created_at":"2026-07-05T04:22:13.344095+00:00"},{"alias_kind":"arxiv_version","alias_value":"2205.05050v1","created_at":"2026-07-05T04:22:13.344095+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2205.05050","created_at":"2026-07-05T04:22:13.344095+00:00"},{"alias_kind":"pith_short_12","alias_value":"YBEWXHEXQIRD","created_at":"2026-07-05T04:22:13.344095+00:00"},{"alias_kind":"pith_short_16","alias_value":"YBEWXHEXQIRDZYXR","created_at":"2026-07-05T04:22:13.344095+00:00"},{"alias_kind":"pith_short_8","alias_value":"YBEWXHEX","created_at":"2026-07-05T04:22:13.344095+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YBEWXHEXQIRDZYXRDD6PVQ77GD","json":"https://pith.science/pith/YBEWXHEXQIRDZYXRDD6PVQ77GD.json","graph_json":"https://pith.science/api/pith-number/YBEWXHEXQIRDZYXRDD6PVQ77GD/graph.json","events_json":"https://pith.science/api/pith-number/YBEWXHEXQIRDZYXRDD6PVQ77GD/events.json","paper":"https://pith.science/paper/YBEWXHEX"},"agent_actions":{"view_html":"https://pith.science/pith/YBEWXHEXQIRDZYXRDD6PVQ77GD","download_json":"https://pith.science/pith/YBEWXHEXQIRDZYXRDD6PVQ77GD.json","view_paper":"https://pith.science/paper/YBEWXHEX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2205.05050&json=true","fetch_graph":"https://pith.science/api/pith-number/YBEWXHEXQIRDZYXRDD6PVQ77GD/graph.json","fetch_events":"https://pith.science/api/pith-number/YBEWXHEXQIRDZYXRDD6PVQ77GD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YBEWXHEXQIRDZYXRDD6PVQ77GD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YBEWXHEXQIRDZYXRDD6PVQ77GD/action/storage_attestation","attest_author":"https://pith.science/pith/YBEWXHEXQIRDZYXRDD6PVQ77GD/action/author_attestation","sign_citation":"https://pith.science/pith/YBEWXHEXQIRDZYXRDD6PVQ77GD/action/citation_signature","submit_replication":"https://pith.science/pith/YBEWXHEXQIRDZYXRDD6PVQ77GD/action/replication_record"}},"created_at":"2026-07-05T04:22:13.344095+00:00","updated_at":"2026-07-05T04:22:13.344095+00:00"}