{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:455NGT2NGRVMK74FVYSOKCSTED","short_pith_number":"pith:455NGT2N","schema_version":"1.0","canonical_sha256":"e77ad34f4d346ac57f85ae24e50a5320f1b2d385cf905a5724acfb8f4e18961a","source":{"kind":"arxiv","id":"2302.12006","version":1},"attestation_state":"computed","paper":{"title":"Does the evaluation stand up to evaluation? A first-principle approach to the evaluation of classifiers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["physics.data-an"],"primary_cat":"cs.LG","authors_text":"A. S. Lundervold, K. Dyrland, P.G.L. Porta Mana","submitted_at":"2023-02-21T09:55:19Z","abstract_excerpt":"How can one meaningfully make a measurement, if the meter does not conform to any standard and its scale expands or shrinks depending on what is measured? In the present work it is argued that current evaluation practices for machine-learning classifiers are affected by this kind of problem, leading to negative consequences when classifiers are put to real use; consequences that could have been avoided. It is proposed that evaluation be grounded on Decision Theory, and the implications of such foundation are explored. The main result is that every evaluation metric must be a linear combination"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2302.12006","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-02-21T09:55:19Z","cross_cats_sorted":["physics.data-an"],"title_canon_sha256":"19ada7f645cb7d796b717c369c95a374ee51e9e47982a3662ca3eead8780763a","abstract_canon_sha256":"42f1e5d969a4a0cce8eba46272e40b557d6c8b44a3f6ec08628a30340c19b2c7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:44:58.915106Z","signature_b64":"xQ1obtZUYsf56qSm3FvK6b+F2VSchKMxzQlj6oj0wUUlcpQuSrja/5mj83+JANoD6sd1hvdiNfUWR6OWIgSOCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e77ad34f4d346ac57f85ae24e50a5320f1b2d385cf905a5724acfb8f4e18961a","last_reissued_at":"2026-07-05T05:44:58.914712Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:44:58.914712Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Does the evaluation stand up to evaluation? A first-principle approach to the evaluation of classifiers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["physics.data-an"],"primary_cat":"cs.LG","authors_text":"A. S. Lundervold, K. Dyrland, P.G.L. Porta Mana","submitted_at":"2023-02-21T09:55:19Z","abstract_excerpt":"How can one meaningfully make a measurement, if the meter does not conform to any standard and its scale expands or shrinks depending on what is measured? In the present work it is argued that current evaluation practices for machine-learning classifiers are affected by this kind of problem, leading to negative consequences when classifiers are put to real use; consequences that could have been avoided. It is proposed that evaluation be grounded on Decision Theory, and the implications of such foundation are explored. The main result is that every evaluation metric must be a linear combination"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.12006","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.12006/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2302.12006","created_at":"2026-07-05T05:44:58.914770+00:00"},{"alias_kind":"arxiv_version","alias_value":"2302.12006v1","created_at":"2026-07-05T05:44:58.914770+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.12006","created_at":"2026-07-05T05:44:58.914770+00:00"},{"alias_kind":"pith_short_12","alias_value":"455NGT2NGRVM","created_at":"2026-07-05T05:44:58.914770+00:00"},{"alias_kind":"pith_short_16","alias_value":"455NGT2NGRVMK74F","created_at":"2026-07-05T05:44:58.914770+00:00"},{"alias_kind":"pith_short_8","alias_value":"455NGT2N","created_at":"2026-07-05T05:44:58.914770+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.20490","citing_title":"ECUAS$_n$: A family of metrics for principled evaluation of uncertainty-augmented systems","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20490","citing_title":"ECUAS$_n$: A family of metrics for principled evaluation of uncertainty-augmented systems","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20490","citing_title":"ECUAS$_n$: A family of metrics for principled evaluation of uncertainty-augmented systems","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/455NGT2NGRVMK74FVYSOKCSTED","json":"https://pith.science/pith/455NGT2NGRVMK74FVYSOKCSTED.json","graph_json":"https://pith.science/api/pith-number/455NGT2NGRVMK74FVYSOKCSTED/graph.json","events_json":"https://pith.science/api/pith-number/455NGT2NGRVMK74FVYSOKCSTED/events.json","paper":"https://pith.science/paper/455NGT2N"},"agent_actions":{"view_html":"https://pith.science/pith/455NGT2NGRVMK74FVYSOKCSTED","download_json":"https://pith.science/pith/455NGT2NGRVMK74FVYSOKCSTED.json","view_paper":"https://pith.science/paper/455NGT2N","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2302.12006&json=true","fetch_graph":"https://pith.science/api/pith-number/455NGT2NGRVMK74FVYSOKCSTED/graph.json","fetch_events":"https://pith.science/api/pith-number/455NGT2NGRVMK74FVYSOKCSTED/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/455NGT2NGRVMK74FVYSOKCSTED/action/timestamp_anchor","attest_storage":"https://pith.science/pith/455NGT2NGRVMK74FVYSOKCSTED/action/storage_attestation","attest_author":"https://pith.science/pith/455NGT2NGRVMK74FVYSOKCSTED/action/author_attestation","sign_citation":"https://pith.science/pith/455NGT2NGRVMK74FVYSOKCSTED/action/citation_signature","submit_replication":"https://pith.science/pith/455NGT2NGRVMK74FVYSOKCSTED/action/replication_record"}},"created_at":"2026-07-05T05:44:58.914770+00:00","updated_at":"2026-07-05T05:44:58.914770+00:00"}