{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:7WOHODA7X536P6KCJSB4ZO65V6","short_pith_number":"pith:7WOHODA7","schema_version":"1.0","canonical_sha256":"fd9c770c1fbf77e7f9424c83ccbbddafb73308b6352cc259ad22ce3acfe30f93","source":{"kind":"arxiv","id":"2406.08723","version":1},"attestation_state":"computed","paper":{"title":"ECBD: Evidence-Centered Benchmark Design for NLP","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alexandra Olteanu, Jackie Chi Kit Cheung, Q. Vera Liao, Su Lin Blodgett, Yu Lu Liu, Ziang Xiao","submitted_at":"2024-06-13T00:59:55Z","abstract_excerpt":"Benchmarking is seen as critical to assessing progress in NLP. However, creating a benchmark involves many design decisions (e.g., which datasets to include, which metrics to use) that often rely on tacit, untested assumptions about what the benchmark is intended to measure or is actually measuring. There is currently no principled way of analyzing these decisions and how they impact the validity of the benchmark's measurements. To address this gap, we draw on evidence-centered design in educational assessments and propose Evidence-Centered Benchmark Design (ECBD), a framework which formalizes"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.08723","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-06-13T00:59:55Z","cross_cats_sorted":[],"title_canon_sha256":"e6caeb02401b3f2de3a35ad503460c5ce711cc3a8298dc1b102482fe7dce9a43","abstract_canon_sha256":"4da76a29463b1203971d2c22b45a8c5455f4aacfb5abad7abf05f9b9ff98450d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:31:22.188938Z","signature_b64":"9bj/EOkQpGa3N6tmXGKAMdtUW+zzbNqwJHuUdZvpI3V8rDmr1r6h5UTG/URvTQ5Vnu+qnSvyVZwp7zH6Q8COAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fd9c770c1fbf77e7f9424c83ccbbddafb73308b6352cc259ad22ce3acfe30f93","last_reissued_at":"2026-07-05T08:31:22.188467Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:31:22.188467Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ECBD: Evidence-Centered Benchmark Design for NLP","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alexandra Olteanu, Jackie Chi Kit Cheung, Q. Vera Liao, Su Lin Blodgett, Yu Lu Liu, Ziang Xiao","submitted_at":"2024-06-13T00:59:55Z","abstract_excerpt":"Benchmarking is seen as critical to assessing progress in NLP. However, creating a benchmark involves many design decisions (e.g., which datasets to include, which metrics to use) that often rely on tacit, untested assumptions about what the benchmark is intended to measure or is actually measuring. There is currently no principled way of analyzing these decisions and how they impact the validity of the benchmark's measurements. To address this gap, we draw on evidence-centered design in educational assessments and propose Evidence-Centered Benchmark Design (ECBD), a framework which formalizes"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.08723","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.08723/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.08723","created_at":"2026-07-05T08:31:22.188519+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.08723v1","created_at":"2026-07-05T08:31:22.188519+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.08723","created_at":"2026-07-05T08:31:22.188519+00:00"},{"alias_kind":"pith_short_12","alias_value":"7WOHODA7X536","created_at":"2026-07-05T08:31:22.188519+00:00"},{"alias_kind":"pith_short_16","alias_value":"7WOHODA7X536P6KC","created_at":"2026-07-05T08:31:22.188519+00:00"},{"alias_kind":"pith_short_8","alias_value":"7WOHODA7","created_at":"2026-07-05T08:31:22.188519+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2511.05501","citing_title":"Towards Real-World Validity in Generative AI Benchmarks: Understanding and Designing Domain-Centered Evaluations for Journalism Practitioners","ref_index":39,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7WOHODA7X536P6KCJSB4ZO65V6","json":"https://pith.science/pith/7WOHODA7X536P6KCJSB4ZO65V6.json","graph_json":"https://pith.science/api/pith-number/7WOHODA7X536P6KCJSB4ZO65V6/graph.json","events_json":"https://pith.science/api/pith-number/7WOHODA7X536P6KCJSB4ZO65V6/events.json","paper":"https://pith.science/paper/7WOHODA7"},"agent_actions":{"view_html":"https://pith.science/pith/7WOHODA7X536P6KCJSB4ZO65V6","download_json":"https://pith.science/pith/7WOHODA7X536P6KCJSB4ZO65V6.json","view_paper":"https://pith.science/paper/7WOHODA7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.08723&json=true","fetch_graph":"https://pith.science/api/pith-number/7WOHODA7X536P6KCJSB4ZO65V6/graph.json","fetch_events":"https://pith.science/api/pith-number/7WOHODA7X536P6KCJSB4ZO65V6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7WOHODA7X536P6KCJSB4ZO65V6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7WOHODA7X536P6KCJSB4ZO65V6/action/storage_attestation","attest_author":"https://pith.science/pith/7WOHODA7X536P6KCJSB4ZO65V6/action/author_attestation","sign_citation":"https://pith.science/pith/7WOHODA7X536P6KCJSB4ZO65V6/action/citation_signature","submit_replication":"https://pith.science/pith/7WOHODA7X536P6KCJSB4ZO65V6/action/replication_record"}},"created_at":"2026-07-05T08:31:22.188519+00:00","updated_at":"2026-07-05T08:31:22.188519+00:00"}