{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:5JAIX3ZMLKC3V6POGI6BVPAJ3D","short_pith_number":"pith:5JAIX3ZM","schema_version":"1.0","canonical_sha256":"ea408bef2c5a85baf9ee323c1abc09d8d989db6c26e13140388a869824caba15","source":{"kind":"arxiv","id":"2104.14337","version":1},"attestation_state":"computed","paper":{"title":"Dynabench: Rethinking Benchmarking in NLP","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Adina Williams, Amanpreet Singh, Atticus Geiger, Bertie Vidgen, Christopher Potts, Divyansh Kaushik, Douwe Kiela, Grusha Prasad, Max Bartolo, Mohit Bansal, Pontus Stenetorp, Pratik Ringshia, Robin Jia, Sebastian Riedel, Tristan Thrush, Yixin Nie, Zeerak Waseem, Zhengxuan Wu, Zhiyi Ma","submitted_at":"2021-04-07T17:49:17Z","abstract_excerpt":"We introduce Dynabench, an open-source platform for dynamic dataset creation and model benchmarking. Dynabench runs in a web browser and supports human-and-model-in-the-loop dataset creation: annotators seek to create examples that a target model will misclassify, but that another person will not. In this paper, we argue that Dynabench addresses a critical need in our community: contemporary models quickly achieve outstanding performance on benchmark tasks but nonetheless fail on simple challenge examples and falter in real-world scenarios. With Dynabench, dataset creation, model development, "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2104.14337","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-04-07T17:49:17Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a60eebbdeedb78e3a6421b8e70cdd26279b1dfe4e7c8b185b7095c20d541680a","abstract_canon_sha256":"91e1e43c0e7aac90b20ca2651ee973a889c85830dd1dc64403a703c0a08d561b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:36:15.475983Z","signature_b64":"Jjfhv6cY6Qr/0oQD3VgJ+GBaoT85OB2bMk1UIosz3ALvyadWS6rk3K3xJSOsu0D1x6uWLx42mf4av1RGKsjeAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ea408bef2c5a85baf9ee323c1abc09d8d989db6c26e13140388a869824caba15","last_reissued_at":"2026-07-05T02:36:15.475559Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:36:15.475559Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dynabench: Rethinking Benchmarking in NLP","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Adina Williams, Amanpreet Singh, Atticus Geiger, Bertie Vidgen, Christopher Potts, Divyansh Kaushik, Douwe Kiela, Grusha Prasad, Max Bartolo, Mohit Bansal, Pontus Stenetorp, Pratik Ringshia, Robin Jia, Sebastian Riedel, Tristan Thrush, Yixin Nie, Zeerak Waseem, Zhengxuan Wu, Zhiyi Ma","submitted_at":"2021-04-07T17:49:17Z","abstract_excerpt":"We introduce Dynabench, an open-source platform for dynamic dataset creation and model benchmarking. Dynabench runs in a web browser and supports human-and-model-in-the-loop dataset creation: annotators seek to create examples that a target model will misclassify, but that another person will not. In this paper, we argue that Dynabench addresses a critical need in our community: contemporary models quickly achieve outstanding performance on benchmark tasks but nonetheless fail on simple challenge examples and falter in real-world scenarios. With Dynabench, dataset creation, model development, "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2104.14337","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2104.14337/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2104.14337","created_at":"2026-07-05T02:36:15.475614+00:00"},{"alias_kind":"arxiv_version","alias_value":"2104.14337v1","created_at":"2026-07-05T02:36:15.475614+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2104.14337","created_at":"2026-07-05T02:36:15.475614+00:00"},{"alias_kind":"pith_short_12","alias_value":"5JAIX3ZMLKC3","created_at":"2026-07-05T02:36:15.475614+00:00"},{"alias_kind":"pith_short_16","alias_value":"5JAIX3ZMLKC3V6PO","created_at":"2026-07-05T02:36:15.475614+00:00"},{"alias_kind":"pith_short_8","alias_value":"5JAIX3ZM","created_at":"2026-07-05T02:36:15.475614+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01740","citing_title":"Meta-Benchmarks for Financial-Services LLM Evaluation","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03650","citing_title":"CoEval: Ranking Language Models for Custom Tasks Without Labeled Data or Trustworthy Benchmarks","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2603.14987","citing_title":"Beyond Benchmark Islands: Toward Representative Trustworthiness Evaluation for Agentic AI","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20520","citing_title":"Open-World Evaluations for Measuring Frontier AI Capabilities","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17829","citing_title":"Interactive Evaluation Requires a Design Science","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2507.22359","citing_title":"League of LLMs: A Benchmark-Free Paradigm for Mutual Evaluation of Large Language Models","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2510.09275","citing_title":"Inflated Excellence or True Performance? Rethinking Medical Diagnostic Benchmarks with Dynamic Evaluation","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10639","citing_title":"Navigating the Sea of LLM Evaluation: Investigating Bias in Toxicity Benchmarks","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04312","citing_title":"Agent Island: A Saturation- and Contamination-Resistant Benchmark from Multiagent Games","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02930","citing_title":"Analysis and Explainability of LLMs Via Evolutionary Methods","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17842","citing_title":"QuickScope: Certifying Hard Questions in Dynamic LLM Benchmarks","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07593","citing_title":"Too long; didn't solve","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05226","citing_title":"RoboPlayground: Democratizing Robotic Evaluation through Structured Physical Domains","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2501.14249","citing_title":"Humanity's Last Exam","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16742","citing_title":"CT Open: An Open-Access, Uncontaminated, Live Platform for the Open Challenge of Clinical Trial Outcome Prediction","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00907","citing_title":"TRIP-Evaluate: An Open Multimodal Benchmark for Evaluating Large Models in Transportation","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5JAIX3ZMLKC3V6POGI6BVPAJ3D","json":"https://pith.science/pith/5JAIX3ZMLKC3V6POGI6BVPAJ3D.json","graph_json":"https://pith.science/api/pith-number/5JAIX3ZMLKC3V6POGI6BVPAJ3D/graph.json","events_json":"https://pith.science/api/pith-number/5JAIX3ZMLKC3V6POGI6BVPAJ3D/events.json","paper":"https://pith.science/paper/5JAIX3ZM"},"agent_actions":{"view_html":"https://pith.science/pith/5JAIX3ZMLKC3V6POGI6BVPAJ3D","download_json":"https://pith.science/pith/5JAIX3ZMLKC3V6POGI6BVPAJ3D.json","view_paper":"https://pith.science/paper/5JAIX3ZM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2104.14337&json=true","fetch_graph":"https://pith.science/api/pith-number/5JAIX3ZMLKC3V6POGI6BVPAJ3D/graph.json","fetch_events":"https://pith.science/api/pith-number/5JAIX3ZMLKC3V6POGI6BVPAJ3D/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5JAIX3ZMLKC3V6POGI6BVPAJ3D/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5JAIX3ZMLKC3V6POGI6BVPAJ3D/action/storage_attestation","attest_author":"https://pith.science/pith/5JAIX3ZMLKC3V6POGI6BVPAJ3D/action/author_attestation","sign_citation":"https://pith.science/pith/5JAIX3ZMLKC3V6POGI6BVPAJ3D/action/citation_signature","submit_replication":"https://pith.science/pith/5JAIX3ZMLKC3V6POGI6BVPAJ3D/action/replication_record"}},"created_at":"2026-07-05T02:36:15.475614+00:00","updated_at":"2026-07-05T02:36:15.475614+00:00"}