{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5NGGKRIJELMSTMGM3FZXVUWMZM","short_pith_number":"pith:5NGGKRIJ","schema_version":"1.0","canonical_sha256":"eb4c65450922d929b0ccd9737ad2cccb0e07ae28e9ac8702f13dedf5292dba6d","source":{"kind":"arxiv","id":"2410.24100","version":1},"attestation_state":"computed","paper":{"title":"Benchmark Data Repositories for Better Benchmarking","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.DL"],"primary_cat":"cs.LG","authors_text":"Markelle Kelly, Padhraic Smyth, Rachel Longjohn, Sameer Singh","submitted_at":"2024-10-31T16:30:08Z","abstract_excerpt":"In machine learning research, it is common to evaluate algorithms via their performance on standard benchmark datasets. While a growing body of work establishes guidelines for -- and levies criticisms at -- data and benchmarking practices in machine learning, comparatively less attention has been paid to the data repositories where these datasets are stored, documented, and shared. In this paper, we analyze the landscape of these $\\textit{benchmark data repositories}$ and the role they can play in improving benchmarking. This role includes addressing issues with both datasets themselves (e.g.,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.24100","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-31T16:30:08Z","cross_cats_sorted":["cs.DL"],"title_canon_sha256":"202271ae43bdf6a1a5b6c5c1ede9a0fde86cbf0866683b6483c6e9c6efaa8499","abstract_canon_sha256":"a4482510c4dd6e73b7362aee99cf372e2f3507bec0c6db0f57e98522b2a768e3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:29:19.953130Z","signature_b64":"UubquvxFdZtMlHhJjfa6oLhyMwnadf0X6G0+08ETpXHy0EWTiA1hYNzJC9x+DQ2D3UKiarLgCFU3ygP5VlAeBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"eb4c65450922d929b0ccd9737ad2cccb0e07ae28e9ac8702f13dedf5292dba6d","last_reissued_at":"2026-07-05T09:29:19.952656Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:29:19.952656Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Benchmark Data Repositories for Better Benchmarking","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.DL"],"primary_cat":"cs.LG","authors_text":"Markelle Kelly, Padhraic Smyth, Rachel Longjohn, Sameer Singh","submitted_at":"2024-10-31T16:30:08Z","abstract_excerpt":"In machine learning research, it is common to evaluate algorithms via their performance on standard benchmark datasets. While a growing body of work establishes guidelines for -- and levies criticisms at -- data and benchmarking practices in machine learning, comparatively less attention has been paid to the data repositories where these datasets are stored, documented, and shared. In this paper, we analyze the landscape of these $\\textit{benchmark data repositories}$ and the role they can play in improving benchmarking. This role includes addressing issues with both datasets themselves (e.g.,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.24100","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.24100/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.24100","created_at":"2026-07-05T09:29:19.952708+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.24100v1","created_at":"2026-07-05T09:29:19.952708+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.24100","created_at":"2026-07-05T09:29:19.952708+00:00"},{"alias_kind":"pith_short_12","alias_value":"5NGGKRIJELMS","created_at":"2026-07-05T09:29:19.952708+00:00"},{"alias_kind":"pith_short_16","alias_value":"5NGGKRIJELMSTMGM","created_at":"2026-07-05T09:29:19.952708+00:00"},{"alias_kind":"pith_short_8","alias_value":"5NGGKRIJ","created_at":"2026-07-05T09:29:19.952708+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.02258","citing_title":"Matter to Mechanism: A Benchmark for AI Co-Scientists in Materials and Battery Research","ref_index":53,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5NGGKRIJELMSTMGM3FZXVUWMZM","json":"https://pith.science/pith/5NGGKRIJELMSTMGM3FZXVUWMZM.json","graph_json":"https://pith.science/api/pith-number/5NGGKRIJELMSTMGM3FZXVUWMZM/graph.json","events_json":"https://pith.science/api/pith-number/5NGGKRIJELMSTMGM3FZXVUWMZM/events.json","paper":"https://pith.science/paper/5NGGKRIJ"},"agent_actions":{"view_html":"https://pith.science/pith/5NGGKRIJELMSTMGM3FZXVUWMZM","download_json":"https://pith.science/pith/5NGGKRIJELMSTMGM3FZXVUWMZM.json","view_paper":"https://pith.science/paper/5NGGKRIJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.24100&json=true","fetch_graph":"https://pith.science/api/pith-number/5NGGKRIJELMSTMGM3FZXVUWMZM/graph.json","fetch_events":"https://pith.science/api/pith-number/5NGGKRIJELMSTMGM3FZXVUWMZM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5NGGKRIJELMSTMGM3FZXVUWMZM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5NGGKRIJELMSTMGM3FZXVUWMZM/action/storage_attestation","attest_author":"https://pith.science/pith/5NGGKRIJELMSTMGM3FZXVUWMZM/action/author_attestation","sign_citation":"https://pith.science/pith/5NGGKRIJELMSTMGM3FZXVUWMZM/action/citation_signature","submit_replication":"https://pith.science/pith/5NGGKRIJELMSTMGM3FZXVUWMZM/action/replication_record"}},"created_at":"2026-07-05T09:29:19.952708+00:00","updated_at":"2026-07-05T09:29:19.952708+00:00"}