{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:AMJ37KAYRMDXB4W5U3NTU6UJGD","short_pith_number":"pith:AMJ37KAY","schema_version":"1.0","canonical_sha256":"0313bfa8188b0770f2dda6db3a7a8930d601138de1f2703ffba05a5877d03dde","source":{"kind":"arxiv","id":"2005.04118","version":1},"attestation_state":"computed","paper":{"title":"Beyond Accuracy: Behavioral Testing of NLP models with CheckList","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Carlos Guestrin, Marco Tulio Ribeiro, Sameer Singh, Tongshuang Wu","submitted_at":"2020-05-08T15:48:31Z","abstract_excerpt":"Although measuring held-out accuracy has been the primary approach to evaluate generalization, it often overestimates the performance of NLP models, while alternative approaches for evaluating models either focus on individual tasks or on specific behaviors. Inspired by principles of behavioral testing in software engineering, we introduce CheckList, a task-agnostic methodology for testing NLP models. CheckList includes a matrix of general linguistic capabilities and test types that facilitate comprehensive test ideation, as well as a software tool to generate a large and diverse number of tes"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2005.04118","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-05-08T15:48:31Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"11fec14a977e09eda3e96c43356cb5f06cdef062c70bc7eb7476736c2c742948","abstract_canon_sha256":"5cbb5295e76de7ee55556cea6316f696fbedd79380f3e7608c7123eb6c670d72"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:01:29.720289Z","signature_b64":"X7kyfgGokSK4nlmMLYG6FQo6zkEPqUPmt9wNkWIPqFpIksnHLMXb005teNfu7LbYEWwUb6TmMC6CHCd0fEDCBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0313bfa8188b0770f2dda6db3a7a8930d601138de1f2703ffba05a5877d03dde","last_reissued_at":"2026-07-05T01:01:29.719808Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:01:29.719808Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Beyond Accuracy: Behavioral Testing of NLP models with CheckList","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Carlos Guestrin, Marco Tulio Ribeiro, Sameer Singh, Tongshuang Wu","submitted_at":"2020-05-08T15:48:31Z","abstract_excerpt":"Although measuring held-out accuracy has been the primary approach to evaluate generalization, it often overestimates the performance of NLP models, while alternative approaches for evaluating models either focus on individual tasks or on specific behaviors. Inspired by principles of behavioral testing in software engineering, we introduce CheckList, a task-agnostic methodology for testing NLP models. CheckList includes a matrix of general linguistic capabilities and test types that facilitate comprehensive test ideation, as well as a software tool to generate a large and diverse number of tes"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2005.04118","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2005.04118/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2005.04118","created_at":"2026-07-05T01:01:29.719867+00:00"},{"alias_kind":"arxiv_version","alias_value":"2005.04118v1","created_at":"2026-07-05T01:01:29.719867+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2005.04118","created_at":"2026-07-05T01:01:29.719867+00:00"},{"alias_kind":"pith_short_12","alias_value":"AMJ37KAYRMDX","created_at":"2026-07-05T01:01:29.719867+00:00"},{"alias_kind":"pith_short_16","alias_value":"AMJ37KAYRMDXB4W5","created_at":"2026-07-05T01:01:29.719867+00:00"},{"alias_kind":"pith_short_8","alias_value":"AMJ37KAY","created_at":"2026-07-05T01:01:29.719867+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19057","citing_title":"Quantifying and Auditing LLM Evaluation via Positive--Unlabeled Learning","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12924","citing_title":"Iterating Toward Better Search: A Two-Agent Simulation Framework for Evaluating Agentic Search Architectures in E-Commerce","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09700","citing_title":"What the Eyes See, the LLMs Miss: Exploiting Human Perception for Adversarial Text Attacks","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04661","citing_title":"CRAFT: Cost-aware Refinement And Front-aware Tuning of Prompts","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28360","citing_title":"Carolina Guide: A Multi-Agent RAG System with Institutional Guardrails for Academic Policy Assistance","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00540","citing_title":"Trustworthy Recommendation in the Era of Large Language Models: Opportunities and Challenges","ref_index":195,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17829","citing_title":"Interactive Evaluation Requires a Design Science","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2509.11206","citing_title":"Evalet: Evaluating Large Language Models through Functional Fragmentation","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2511.01458","citing_title":"When to Trust the Answer: Question-Aligned Semantic Nearest Neighbor Entropy for Safer Surgical VQA","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2601.16175","citing_title":"Learning to Discover at Test Time","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13625","citing_title":"How to Interpret Agent Behavior","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16421","citing_title":"Measuring Representation Robustness in Large Language Models for Geometry","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2310.08419","citing_title":"Jailbreaking Black Box Large Language Models in Twenty Queries","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00382","citing_title":"Social Bias in LLM-Generated Code: Benchmark and Mitigation","ref_index":153,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AMJ37KAYRMDXB4W5U3NTU6UJGD","json":"https://pith.science/pith/AMJ37KAYRMDXB4W5U3NTU6UJGD.json","graph_json":"https://pith.science/api/pith-number/AMJ37KAYRMDXB4W5U3NTU6UJGD/graph.json","events_json":"https://pith.science/api/pith-number/AMJ37KAYRMDXB4W5U3NTU6UJGD/events.json","paper":"https://pith.science/paper/AMJ37KAY"},"agent_actions":{"view_html":"https://pith.science/pith/AMJ37KAYRMDXB4W5U3NTU6UJGD","download_json":"https://pith.science/pith/AMJ37KAYRMDXB4W5U3NTU6UJGD.json","view_paper":"https://pith.science/paper/AMJ37KAY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2005.04118&json=true","fetch_graph":"https://pith.science/api/pith-number/AMJ37KAYRMDXB4W5U3NTU6UJGD/graph.json","fetch_events":"https://pith.science/api/pith-number/AMJ37KAYRMDXB4W5U3NTU6UJGD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AMJ37KAYRMDXB4W5U3NTU6UJGD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AMJ37KAYRMDXB4W5U3NTU6UJGD/action/storage_attestation","attest_author":"https://pith.science/pith/AMJ37KAYRMDXB4W5U3NTU6UJGD/action/author_attestation","sign_citation":"https://pith.science/pith/AMJ37KAYRMDXB4W5U3NTU6UJGD/action/citation_signature","submit_replication":"https://pith.science/pith/AMJ37KAYRMDXB4W5U3NTU6UJGD/action/replication_record"}},"created_at":"2026-07-05T01:01:29.719867+00:00","updated_at":"2026-07-05T01:01:29.719867+00:00"}