{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:I7KLMSG442SAB5GFOH5EG7CIE2","short_pith_number":"pith:I7KLMSG4","schema_version":"1.0","canonical_sha256":"47d4b648dce6a400f4c571fa437c4826a6ba324d1a935262174394b7289de006","source":{"kind":"arxiv","id":"2205.11097","version":2},"attestation_state":"computed","paper":{"title":"A Fine-grained Interpretability Evaluation Benchmark for Neural NLP","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"HaiFeng Wang, Hao Liu, Hongxuan Tang, Hua Wu, Lijie Wang, Shuai Zhang, Shuyuan Peng, Xinyan Xiao, Yaozong Shen, Ying Chen","submitted_at":"2022-05-23T07:37:04Z","abstract_excerpt":"While there is increasing concern about the interpretability of neural models, the evaluation of interpretability remains an open problem, due to the lack of proper evaluation datasets and metrics. In this paper, we present a novel benchmark to evaluate the interpretability of both neural models and saliency methods. This benchmark covers three representative NLP tasks: sentiment analysis, textual similarity and reading comprehension, each provided with both English and Chinese annotated data. In order to precisely evaluate the interpretability, we provide token-level rationales that are caref"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2205.11097","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2022-05-23T07:37:04Z","cross_cats_sorted":[],"title_canon_sha256":"20ead4a0d984fb8944234cb7edbf0122429f35f8d2986c3899ed0b906f51c1d3","abstract_canon_sha256":"440529be0c132b521bfe1e30028298f7d6f9f18e850c8c03d42aafb4203bc164"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:16:00.475087Z","signature_b64":"ioI9RJRwsPzDM2pPeqiyKiDIIIdaNVW4AnuifBNn9gJTpanZbYo3d7ScKfbSnLhujDdjDyUWtfdrFJuFu6xcAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"47d4b648dce6a400f4c571fa437c4826a6ba324d1a935262174394b7289de006","last_reissued_at":"2026-07-05T05:16:00.474625Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:16:00.474625Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Fine-grained Interpretability Evaluation Benchmark for Neural NLP","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"HaiFeng Wang, Hao Liu, Hongxuan Tang, Hua Wu, Lijie Wang, Shuai Zhang, Shuyuan Peng, Xinyan Xiao, Yaozong Shen, Ying Chen","submitted_at":"2022-05-23T07:37:04Z","abstract_excerpt":"While there is increasing concern about the interpretability of neural models, the evaluation of interpretability remains an open problem, due to the lack of proper evaluation datasets and metrics. In this paper, we present a novel benchmark to evaluate the interpretability of both neural models and saliency methods. This benchmark covers three representative NLP tasks: sentiment analysis, textual similarity and reading comprehension, each provided with both English and Chinese annotated data. In order to precisely evaluate the interpretability, we provide token-level rationales that are caref"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2205.11097","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2205.11097/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2205.11097","created_at":"2026-07-05T05:16:00.474683+00:00"},{"alias_kind":"arxiv_version","alias_value":"2205.11097v2","created_at":"2026-07-05T05:16:00.474683+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2205.11097","created_at":"2026-07-05T05:16:00.474683+00:00"},{"alias_kind":"pith_short_12","alias_value":"I7KLMSG442SA","created_at":"2026-07-05T05:16:00.474683+00:00"},{"alias_kind":"pith_short_16","alias_value":"I7KLMSG442SAB5GF","created_at":"2026-07-05T05:16:00.474683+00:00"},{"alias_kind":"pith_short_8","alias_value":"I7KLMSG4","created_at":"2026-07-05T05:16:00.474683+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/I7KLMSG442SAB5GFOH5EG7CIE2","json":"https://pith.science/pith/I7KLMSG442SAB5GFOH5EG7CIE2.json","graph_json":"https://pith.science/api/pith-number/I7KLMSG442SAB5GFOH5EG7CIE2/graph.json","events_json":"https://pith.science/api/pith-number/I7KLMSG442SAB5GFOH5EG7CIE2/events.json","paper":"https://pith.science/paper/I7KLMSG4"},"agent_actions":{"view_html":"https://pith.science/pith/I7KLMSG442SAB5GFOH5EG7CIE2","download_json":"https://pith.science/pith/I7KLMSG442SAB5GFOH5EG7CIE2.json","view_paper":"https://pith.science/paper/I7KLMSG4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2205.11097&json=true","fetch_graph":"https://pith.science/api/pith-number/I7KLMSG442SAB5GFOH5EG7CIE2/graph.json","fetch_events":"https://pith.science/api/pith-number/I7KLMSG442SAB5GFOH5EG7CIE2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/I7KLMSG442SAB5GFOH5EG7CIE2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/I7KLMSG442SAB5GFOH5EG7CIE2/action/storage_attestation","attest_author":"https://pith.science/pith/I7KLMSG442SAB5GFOH5EG7CIE2/action/author_attestation","sign_citation":"https://pith.science/pith/I7KLMSG442SAB5GFOH5EG7CIE2/action/citation_signature","submit_replication":"https://pith.science/pith/I7KLMSG442SAB5GFOH5EG7CIE2/action/replication_record"}},"created_at":"2026-07-05T05:16:00.474683+00:00","updated_at":"2026-07-05T05:16:00.474683+00:00"}