{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:U6EAZM6S33RBD6DWHKMXF3W67F","short_pith_number":"pith:U6EAZM6S","schema_version":"1.0","canonical_sha256":"a7880cb3d2dee211f8763a9972eedef97c78201830a5d6a5cd64d687f44bae2c","source":{"kind":"arxiv","id":"2111.02840","version":2},"attestation_state":"computed","paper":{"title":"Adversarial GLUE: A Multi-Task Benchmark for Robustness Evaluation of Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Ahmed Hassan Awadallah, Bo Li, Boxin Wang, Chejian Xu, Jianfeng Gao, Shuohang Wang, Yu Cheng, Zhe Gan","submitted_at":"2021-11-04T12:59:55Z","abstract_excerpt":"Large-scale pre-trained language models have achieved tremendous success across a wide range of natural language understanding (NLU) tasks, even surpassing human performance. However, recent studies reveal that the robustness of these models can be challenged by carefully crafted textual adversarial examples. While several individual datasets have been proposed to evaluate model robustness, a principled and comprehensive benchmark is still missing. In this paper, we present Adversarial GLUE (AdvGLUE), a new multi-task benchmark to quantitatively and thoroughly explore and evaluate the vulnerab"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2111.02840","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2021-11-04T12:59:55Z","cross_cats_sorted":["cs.CR","cs.LG"],"title_canon_sha256":"800adeae09acf562e047777558566c274369efbaf1a00859d6b20eefcc93b4f7","abstract_canon_sha256":"8d39157ebb41f1f4ac6c46bbfc84bb5a9dbac79be624b839d0b89e614714eb1c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:47:02.194819Z","signature_b64":"+NghZr3IOu+twkijBssjneT7UjMsOFdQHoZ19aChX5I+FOE0TGCi4l11vrd+4JvGlhcIV18LBGON+E0zEBGeBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a7880cb3d2dee211f8763a9972eedef97c78201830a5d6a5cd64d687f44bae2c","last_reissued_at":"2026-07-05T03:47:02.194327Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:47:02.194327Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Adversarial GLUE: A Multi-Task Benchmark for Robustness Evaluation of Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Ahmed Hassan Awadallah, Bo Li, Boxin Wang, Chejian Xu, Jianfeng Gao, Shuohang Wang, Yu Cheng, Zhe Gan","submitted_at":"2021-11-04T12:59:55Z","abstract_excerpt":"Large-scale pre-trained language models have achieved tremendous success across a wide range of natural language understanding (NLU) tasks, even surpassing human performance. However, recent studies reveal that the robustness of these models can be challenged by carefully crafted textual adversarial examples. While several individual datasets have been proposed to evaluate model robustness, a principled and comprehensive benchmark is still missing. In this paper, we present Adversarial GLUE (AdvGLUE), a new multi-task benchmark to quantitatively and thoroughly explore and evaluate the vulnerab"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2111.02840","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2111.02840/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2111.02840","created_at":"2026-07-05T03:47:02.194387+00:00"},{"alias_kind":"arxiv_version","alias_value":"2111.02840v2","created_at":"2026-07-05T03:47:02.194387+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2111.02840","created_at":"2026-07-05T03:47:02.194387+00:00"},{"alias_kind":"pith_short_12","alias_value":"U6EAZM6S33RB","created_at":"2026-07-05T03:47:02.194387+00:00"},{"alias_kind":"pith_short_16","alias_value":"U6EAZM6S33RBD6DW","created_at":"2026-07-05T03:47:02.194387+00:00"},{"alias_kind":"pith_short_8","alias_value":"U6EAZM6S","created_at":"2026-07-05T03:47:02.194387+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09711","citing_title":"Proxy Reward Internalization and Mechanistic Exploitation: A Learned Precursor to Reward Hacking and Its Generalization","ref_index":91,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00012","citing_title":"PRA-RAG: Provably Robust Aggregation in Retrieval-Augmented Generation against Retrieval Corruption","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03606","citing_title":"Testing LLM Arithmetic Reasoning Generalization with Automatic Numeric-Remapping Attacks","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2307.15043","citing_title":"Universal and Transferable Adversarial Attacks on Aligned Language Models","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2406.04244","citing_title":"Benchmark Data Contamination of Large Language Models: A Survey","ref_index":149,"is_internal_anchor":false},{"citing_arxiv_id":"2507.05660","citing_title":"Optimus: A Robust Defense Framework for Mitigating Toxicity while Fine-Tuning Conversational AI","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20654","citing_title":"REFLECTOR: Internalizing Step-wise Reflection against Indirect Jailbreak","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2401.05561","citing_title":"TrustLLM: Trustworthiness in Large Language Models","ref_index":267,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14415","citing_title":"SWE-Chain: Benchmarking Coding Agents on Chained Release-Level Package Upgrades","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18389","citing_title":"Understanding the Prompt Sensitivity","ref_index":64,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/U6EAZM6S33RBD6DWHKMXF3W67F","json":"https://pith.science/pith/U6EAZM6S33RBD6DWHKMXF3W67F.json","graph_json":"https://pith.science/api/pith-number/U6EAZM6S33RBD6DWHKMXF3W67F/graph.json","events_json":"https://pith.science/api/pith-number/U6EAZM6S33RBD6DWHKMXF3W67F/events.json","paper":"https://pith.science/paper/U6EAZM6S"},"agent_actions":{"view_html":"https://pith.science/pith/U6EAZM6S33RBD6DWHKMXF3W67F","download_json":"https://pith.science/pith/U6EAZM6S33RBD6DWHKMXF3W67F.json","view_paper":"https://pith.science/paper/U6EAZM6S","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2111.02840&json=true","fetch_graph":"https://pith.science/api/pith-number/U6EAZM6S33RBD6DWHKMXF3W67F/graph.json","fetch_events":"https://pith.science/api/pith-number/U6EAZM6S33RBD6DWHKMXF3W67F/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/U6EAZM6S33RBD6DWHKMXF3W67F/action/timestamp_anchor","attest_storage":"https://pith.science/pith/U6EAZM6S33RBD6DWHKMXF3W67F/action/storage_attestation","attest_author":"https://pith.science/pith/U6EAZM6S33RBD6DWHKMXF3W67F/action/author_attestation","sign_citation":"https://pith.science/pith/U6EAZM6S33RBD6DWHKMXF3W67F/action/citation_signature","submit_replication":"https://pith.science/pith/U6EAZM6S33RBD6DWHKMXF3W67F/action/replication_record"}},"created_at":"2026-07-05T03:47:02.194387+00:00","updated_at":"2026-07-05T03:47:02.194387+00:00"}