{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:RAWXNO76EEC6PNJQMXFAMR77RL","short_pith_number":"pith:RAWXNO76","schema_version":"1.0","canonical_sha256":"882d76bbfe2105e7b53065ca0647ff8ade42ecc77bd514172f7bcf5d44b345c8","source":{"kind":"arxiv","id":"2106.16020","version":1},"attestation_state":"computed","paper":{"title":"Anomaly Detection: How to Artificially Increase your F1-Score with a Biased Evaluation Protocol","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Damien Fourure, Muhammad Usama Javaid, Nicolas Posocco, Simon Tihon","submitted_at":"2021-06-30T12:36:01Z","abstract_excerpt":"Anomaly detection is a widely explored domain in machine learning. Many models are proposed in the literature, and compared through different metrics measured on various datasets. The most popular metrics used to compare performances are F1-score, AUC and AVPR. In this paper, we show that F1-score and AVPR are highly sensitive to the contamination rate. One consequence is that it is possible to artificially increase their values by modifying the train-test split procedure. This leads to misleading comparisons between algorithms in the literature, especially when the evaluation protocol is not "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.16020","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-06-30T12:36:01Z","cross_cats_sorted":[],"title_canon_sha256":"c2a96e164e83d836bfc99ef9f9dd5634c9ae99f2a2e15ddad03be8d5c042df0f","abstract_canon_sha256":"0b9b67d9f24fcd9b4db52f8f42a36bf05dc6169d7714b6701e60fe27e5a67cdb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:54:02.474737Z","signature_b64":"ACuDnOLUjuxm19aeH+hWEfmVqKZV8rZBBNg6e7l78oH8G68tn27RMHaeiXzV7deWQWSVxtTh3Rys8gwQGSMICQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"882d76bbfe2105e7b53065ca0647ff8ade42ecc77bd514172f7bcf5d44b345c8","last_reissued_at":"2026-07-05T02:54:02.474334Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:54:02.474334Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Anomaly Detection: How to Artificially Increase your F1-Score with a Biased Evaluation Protocol","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Damien Fourure, Muhammad Usama Javaid, Nicolas Posocco, Simon Tihon","submitted_at":"2021-06-30T12:36:01Z","abstract_excerpt":"Anomaly detection is a widely explored domain in machine learning. Many models are proposed in the literature, and compared through different metrics measured on various datasets. The most popular metrics used to compare performances are F1-score, AUC and AVPR. In this paper, we show that F1-score and AVPR are highly sensitive to the contamination rate. One consequence is that it is possible to artificially increase their values by modifying the train-test split procedure. This leads to misleading comparisons between algorithms in the literature, especially when the evaluation protocol is not "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.16020","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.16020/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.16020","created_at":"2026-07-05T02:54:02.474397+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.16020v1","created_at":"2026-07-05T02:54:02.474397+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.16020","created_at":"2026-07-05T02:54:02.474397+00:00"},{"alias_kind":"pith_short_12","alias_value":"RAWXNO76EEC6","created_at":"2026-07-05T02:54:02.474397+00:00"},{"alias_kind":"pith_short_16","alias_value":"RAWXNO76EEC6PNJQ","created_at":"2026-07-05T02:54:02.474397+00:00"},{"alias_kind":"pith_short_8","alias_value":"RAWXNO76","created_at":"2026-07-05T02:54:02.474397+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RAWXNO76EEC6PNJQMXFAMR77RL","json":"https://pith.science/pith/RAWXNO76EEC6PNJQMXFAMR77RL.json","graph_json":"https://pith.science/api/pith-number/RAWXNO76EEC6PNJQMXFAMR77RL/graph.json","events_json":"https://pith.science/api/pith-number/RAWXNO76EEC6PNJQMXFAMR77RL/events.json","paper":"https://pith.science/paper/RAWXNO76"},"agent_actions":{"view_html":"https://pith.science/pith/RAWXNO76EEC6PNJQMXFAMR77RL","download_json":"https://pith.science/pith/RAWXNO76EEC6PNJQMXFAMR77RL.json","view_paper":"https://pith.science/paper/RAWXNO76","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.16020&json=true","fetch_graph":"https://pith.science/api/pith-number/RAWXNO76EEC6PNJQMXFAMR77RL/graph.json","fetch_events":"https://pith.science/api/pith-number/RAWXNO76EEC6PNJQMXFAMR77RL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RAWXNO76EEC6PNJQMXFAMR77RL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RAWXNO76EEC6PNJQMXFAMR77RL/action/storage_attestation","attest_author":"https://pith.science/pith/RAWXNO76EEC6PNJQMXFAMR77RL/action/author_attestation","sign_citation":"https://pith.science/pith/RAWXNO76EEC6PNJQMXFAMR77RL/action/citation_signature","submit_replication":"https://pith.science/pith/RAWXNO76EEC6PNJQMXFAMR77RL/action/replication_record"}},"created_at":"2026-07-05T02:54:02.474397+00:00","updated_at":"2026-07-05T02:54:02.474397+00:00"}