{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:VE6J43XA7IQSYEQM62JKAIJQB3","short_pith_number":"pith:VE6J43XA","schema_version":"1.0","canonical_sha256":"a93c9e6ee0fa212c120cf692a021300ee38a4b042e0acc869f4aa08bef520b56","source":{"kind":"arxiv","id":"2305.11921","version":1},"attestation_state":"computed","paper":{"title":"An Approach to Multiple Comparison Benchmark Evaluations that is Stable Under Manipulation of the Comparate Set","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.PF"],"primary_cat":"stat.ME","authors_text":"Ali Ismail-Fawaz, Angus Dempster, Chang Wei Tan, Daniel F. Schmidt, Geoffrey I. Webb, Germain Forestier, Jonathan Weber, Lynn Miller, Matthieu Herrmann, Maxime Devanne, Stefano Berretti","submitted_at":"2023-05-19T08:58:55Z","abstract_excerpt":"The measurement of progress using benchmarks evaluations is ubiquitous in computer science and machine learning. However, common approaches to analyzing and presenting the results of benchmark comparisons of multiple algorithms over multiple datasets, such as the critical difference diagram introduced by Dem\\v{s}ar (2006), have important shortcomings and, we show, are open to both inadvertent and intentional manipulation. To address these issues, we propose a new approach to presenting the results of benchmark comparisons, the Multiple Comparison Matrix (MCM), that prioritizes pairwise compari"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.11921","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"stat.ME","submitted_at":"2023-05-19T08:58:55Z","cross_cats_sorted":["cs.AI","cs.LG","cs.PF"],"title_canon_sha256":"1931d48336e31832c515875c05d9b5b9f59968c5b7dbb4a878b72fc6b61a5ca4","abstract_canon_sha256":"8493e2bc5c1a70ef8913985e56397dec0f17adb4fad978764691319b940bd533"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:12:04.544209Z","signature_b64":"T3eliVss1dHolPHX5NDIoptr3NGqVt4qtLV0W7L5DXBbd0icVv2UGaS4t7sGk4Pg0NROitGpNpzBmLrqFiO4Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a93c9e6ee0fa212c120cf692a021300ee38a4b042e0acc869f4aa08bef520b56","last_reissued_at":"2026-07-05T06:12:04.543813Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:12:04.543813Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An Approach to Multiple Comparison Benchmark Evaluations that is Stable Under Manipulation of the Comparate Set","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.PF"],"primary_cat":"stat.ME","authors_text":"Ali Ismail-Fawaz, Angus Dempster, Chang Wei Tan, Daniel F. Schmidt, Geoffrey I. Webb, Germain Forestier, Jonathan Weber, Lynn Miller, Matthieu Herrmann, Maxime Devanne, Stefano Berretti","submitted_at":"2023-05-19T08:58:55Z","abstract_excerpt":"The measurement of progress using benchmarks evaluations is ubiquitous in computer science and machine learning. However, common approaches to analyzing and presenting the results of benchmark comparisons of multiple algorithms over multiple datasets, such as the critical difference diagram introduced by Dem\\v{s}ar (2006), have important shortcomings and, we show, are open to both inadvertent and intentional manipulation. To address these issues, we propose a new approach to presenting the results of benchmark comparisons, the Multiple Comparison Matrix (MCM), that prioritizes pairwise compari"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.11921","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.11921/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.11921","created_at":"2026-07-05T06:12:04.543872+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.11921v1","created_at":"2026-07-05T06:12:04.543872+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.11921","created_at":"2026-07-05T06:12:04.543872+00:00"},{"alias_kind":"pith_short_12","alias_value":"VE6J43XA7IQS","created_at":"2026-07-05T06:12:04.543872+00:00"},{"alias_kind":"pith_short_16","alias_value":"VE6J43XA7IQSYEQM","created_at":"2026-07-05T06:12:04.543872+00:00"},{"alias_kind":"pith_short_8","alias_value":"VE6J43XA","created_at":"2026-07-05T06:12:04.543872+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2409.01115","citing_title":"Time series classification with random convolution kernels: pooling operators and input representations matter","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00069","citing_title":"Soft-MSM: Differentiable Context-Aware Elastic Alignment for Time Series","ref_index":192,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VE6J43XA7IQSYEQM62JKAIJQB3","json":"https://pith.science/pith/VE6J43XA7IQSYEQM62JKAIJQB3.json","graph_json":"https://pith.science/api/pith-number/VE6J43XA7IQSYEQM62JKAIJQB3/graph.json","events_json":"https://pith.science/api/pith-number/VE6J43XA7IQSYEQM62JKAIJQB3/events.json","paper":"https://pith.science/paper/VE6J43XA"},"agent_actions":{"view_html":"https://pith.science/pith/VE6J43XA7IQSYEQM62JKAIJQB3","download_json":"https://pith.science/pith/VE6J43XA7IQSYEQM62JKAIJQB3.json","view_paper":"https://pith.science/paper/VE6J43XA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.11921&json=true","fetch_graph":"https://pith.science/api/pith-number/VE6J43XA7IQSYEQM62JKAIJQB3/graph.json","fetch_events":"https://pith.science/api/pith-number/VE6J43XA7IQSYEQM62JKAIJQB3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VE6J43XA7IQSYEQM62JKAIJQB3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VE6J43XA7IQSYEQM62JKAIJQB3/action/storage_attestation","attest_author":"https://pith.science/pith/VE6J43XA7IQSYEQM62JKAIJQB3/action/author_attestation","sign_citation":"https://pith.science/pith/VE6J43XA7IQSYEQM62JKAIJQB3/action/citation_signature","submit_replication":"https://pith.science/pith/VE6J43XA7IQSYEQM62JKAIJQB3/action/replication_record"}},"created_at":"2026-07-05T06:12:04.543872+00:00","updated_at":"2026-07-05T06:12:04.543872+00:00"}