{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KFELEO3LQUUWS4OEUG7UAQC4MO","short_pith_number":"pith:KFELEO3L","schema_version":"1.0","canonical_sha256":"5148b23b6b85296971c4a1bf40405c63a7759d9d66493ee22c37296c3b279e3d","source":{"kind":"arxiv","id":"2410.20245","version":2},"attestation_state":"computed","paper":{"title":"Improving Model Evaluation using SMART Filtering of Benchmark Datasets","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Adina Williams, Candace Ross, David Pantoja, Megan Ung, Rebecca J. Passonneau, Vipul Gupta","submitted_at":"2024-10-26T18:21:44Z","abstract_excerpt":"One of the most challenging problems facing NLP today is evaluation. Some of the most pressing issues pertain to benchmark saturation, data contamination, and diversity in the quality of test examples. To address these concerns, we propose Selection Methodology for Accurate, Reduced, and Targeted (SMART) filtering, a novel approach to select a high-quality subset of examples from existing benchmark datasets by systematically removing less informative and less challenging examples. Our approach applies three filtering criteria, removing (i) easy examples, (ii) data-contaminated examples, and (i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.20245","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-26T18:21:44Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"b0aa3b8ef463d9cd1d2e294f1c686c7ecbb86b389e02135ae34d6f1c23719841","abstract_canon_sha256":"65a46ac953a289da7a67e399c87a5dcdef413213590a3a1cfe83453ee38496fb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:12:22.622194Z","signature_b64":"m0aTYV8IngzLC5xhoI0K1oTrqIC1R+THHb4rgvb1yVjEe1+7H++XQAtXa8gnskxTMoN1I28MwhoRhZcK6fkPBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5148b23b6b85296971c4a1bf40405c63a7759d9d66493ee22c37296c3b279e3d","last_reissued_at":"2026-07-05T10:12:22.621708Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:12:22.621708Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Improving Model Evaluation using SMART Filtering of Benchmark Datasets","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Adina Williams, Candace Ross, David Pantoja, Megan Ung, Rebecca J. Passonneau, Vipul Gupta","submitted_at":"2024-10-26T18:21:44Z","abstract_excerpt":"One of the most challenging problems facing NLP today is evaluation. Some of the most pressing issues pertain to benchmark saturation, data contamination, and diversity in the quality of test examples. To address these concerns, we propose Selection Methodology for Accurate, Reduced, and Targeted (SMART) filtering, a novel approach to select a high-quality subset of examples from existing benchmark datasets by systematically removing less informative and less challenging examples. Our approach applies three filtering criteria, removing (i) easy examples, (ii) data-contaminated examples, and (i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.20245","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.20245/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.20245","created_at":"2026-07-05T10:12:22.621768+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.20245v2","created_at":"2026-07-05T10:12:22.621768+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.20245","created_at":"2026-07-05T10:12:22.621768+00:00"},{"alias_kind":"pith_short_12","alias_value":"KFELEO3LQUUW","created_at":"2026-07-05T10:12:22.621768+00:00"},{"alias_kind":"pith_short_16","alias_value":"KFELEO3LQUUWS4OE","created_at":"2026-07-05T10:12:22.621768+00:00"},{"alias_kind":"pith_short_8","alias_value":"KFELEO3L","created_at":"2026-07-05T10:12:22.621768+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.17747","citing_title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","ref_index":28,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KFELEO3LQUUWS4OEUG7UAQC4MO","json":"https://pith.science/pith/KFELEO3LQUUWS4OEUG7UAQC4MO.json","graph_json":"https://pith.science/api/pith-number/KFELEO3LQUUWS4OEUG7UAQC4MO/graph.json","events_json":"https://pith.science/api/pith-number/KFELEO3LQUUWS4OEUG7UAQC4MO/events.json","paper":"https://pith.science/paper/KFELEO3L"},"agent_actions":{"view_html":"https://pith.science/pith/KFELEO3LQUUWS4OEUG7UAQC4MO","download_json":"https://pith.science/pith/KFELEO3LQUUWS4OEUG7UAQC4MO.json","view_paper":"https://pith.science/paper/KFELEO3L","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.20245&json=true","fetch_graph":"https://pith.science/api/pith-number/KFELEO3LQUUWS4OEUG7UAQC4MO/graph.json","fetch_events":"https://pith.science/api/pith-number/KFELEO3LQUUWS4OEUG7UAQC4MO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KFELEO3LQUUWS4OEUG7UAQC4MO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KFELEO3LQUUWS4OEUG7UAQC4MO/action/storage_attestation","attest_author":"https://pith.science/pith/KFELEO3LQUUWS4OEUG7UAQC4MO/action/author_attestation","sign_citation":"https://pith.science/pith/KFELEO3LQUUWS4OEUG7UAQC4MO/action/citation_signature","submit_replication":"https://pith.science/pith/KFELEO3LQUUWS4OEUG7UAQC4MO/action/replication_record"}},"created_at":"2026-07-05T10:12:22.621768+00:00","updated_at":"2026-07-05T10:12:22.621768+00:00"}