{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:PAEESUG733HZB4UVHFZ3RVSO32","short_pith_number":"pith:PAEESUG7","schema_version":"1.0","canonical_sha256":"78084950dfdecf90f2953973b8d64ede850742075d1722dbd1608a438e30758f","source":{"kind":"arxiv","id":"2604.15038","version":2},"attestation_state":"computed","paper":{"title":"When Fairness Metrics Disagree: Evaluating the Reliability of Demographic Fairness Assessment in Machine Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"Different fairness metrics can lead to contradictory conclusions about bias in the same machine learning model.","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.LG","authors_text":"Khalid Adnan Alsayed","submitted_at":"2026-04-16T14:07:37Z","abstract_excerpt":"The evaluation of fairness in machine learning systems has become a central concern in high-stakes applications, including biometric recognition, healthcare decision-making, and automated risk assessment. Existing approaches typically rely on a small number of fairness metrics to assess model behaviour across group partitions, implicitly assuming that these metrics provide consistent and reliable conclusions. However, different fairness metrics capture distinct statistical properties of model performance and may therefore produce conflicting assessments when applied to the same system. In this"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2604.15038","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-04-16T14:07:37Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"1f81ee62dac09bb08d7d7c8bb74028367860b15e91dc1e267bd4d3eee0abdcc6","abstract_canon_sha256":"f4ca68405cb93eeaef69b4ac82fa8fe476fd5c0ef2c37d6ced48d4291c12093a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-21T01:04:25.831555Z","signature_b64":"mbl+BNgaPfOkLX86i76rXC2Uv1kayOebPnYJ8zCsgYeJZ/rmhGhMf5YKiEl3r5etCBN1fzeOZA8wc2N3kmEPCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"78084950dfdecf90f2953973b8d64ede850742075d1722dbd1608a438e30758f","last_reissued_at":"2026-05-21T01:04:25.830809Z","signature_status":"signed_v1","first_computed_at":"2026-05-21T01:04:25.830809Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"When Fairness Metrics Disagree: Evaluating the Reliability of Demographic Fairness Assessment in Machine Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"Different fairness metrics can lead to contradictory conclusions about bias in the same machine learning model.","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.LG","authors_text":"Khalid Adnan Alsayed","submitted_at":"2026-04-16T14:07:37Z","abstract_excerpt":"The evaluation of fairness in machine learning systems has become a central concern in high-stakes applications, including biometric recognition, healthcare decision-making, and automated risk assessment. Existing approaches typically rely on a small number of fairness metrics to assess model behaviour across group partitions, implicitly assuming that these metrics provide consistent and reliable conclusions. However, different fairness metrics capture distinct statistical properties of model performance and may therefore produce conflicting assessments when applied to the same system. In this"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"Our results demonstrate that fairness assessments can vary significantly depending on the choice of metrics, leading to contradictory conclusions regarding model bias.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That the selected fairness metrics, group partitions, and face-recognition setting are representative enough to generalize the disagreement finding to broader ML fairness evaluation.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"Fairness metrics frequently disagree on bias levels in ML models, quantified by a new Fairness Disagreement Index that remains high across thresholds and configurations.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"Different fairness metrics can lead to contradictory conclusions about bias in the same machine learning model.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"5f527a3a16e6f58b46ccc1ac837291e80648b6ed321df8510b0b7b07dffd2b5a"},"source":{"id":"2604.15038","kind":"arxiv","version":2},"verdict":{"id":"bd085e44-15ec-47a6-85d1-66fa02acb07b","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-10T12:01:20.674846Z","strongest_claim":"Our results demonstrate that fairness assessments can vary significantly depending on the choice of metrics, leading to contradictory conclusions regarding model bias.","one_line_summary":"Fairness metrics frequently disagree on bias levels in ML models, quantified by a new Fairness Disagreement Index that remains high across thresholds and configurations.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That the selected fairness metrics, group partitions, and face-recognition setting are representative enough to generalize the disagreement finding to broader ML fairness evaluation.","pith_extraction_headline":"Different fairness metrics can lead to contradictory conclusions about bias in the same machine learning model."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2604.15038/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2604.15038","created_at":"2026-05-21T01:04:25.830922+00:00"},{"alias_kind":"arxiv_version","alias_value":"2604.15038v2","created_at":"2026-05-21T01:04:25.830922+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2604.15038","created_at":"2026-05-21T01:04:25.830922+00:00"},{"alias_kind":"pith_short_12","alias_value":"PAEESUG733HZ","created_at":"2026-05-21T01:04:25.830922+00:00"},{"alias_kind":"pith_short_16","alias_value":"PAEESUG733HZB4UV","created_at":"2026-05-21T01:04:25.830922+00:00"},{"alias_kind":"pith_short_8","alias_value":"PAEESUG7","created_at":"2026-05-21T01:04:25.830922+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2605.27827","citing_title":"Operational AI Deployment Assurance: Governance-State Orchestration Under Threshold-Sensitive Deployment Conditions -- A Governance Framework for High-Stakes AI Systems","ref_index":12,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PAEESUG733HZB4UVHFZ3RVSO32","json":"https://pith.science/pith/PAEESUG733HZB4UVHFZ3RVSO32.json","graph_json":"https://pith.science/api/pith-number/PAEESUG733HZB4UVHFZ3RVSO32/graph.json","events_json":"https://pith.science/api/pith-number/PAEESUG733HZB4UVHFZ3RVSO32/events.json","paper":"https://pith.science/paper/PAEESUG7"},"agent_actions":{"view_html":"https://pith.science/pith/PAEESUG733HZB4UVHFZ3RVSO32","download_json":"https://pith.science/pith/PAEESUG733HZB4UVHFZ3RVSO32.json","view_paper":"https://pith.science/paper/PAEESUG7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2604.15038&json=true","fetch_graph":"https://pith.science/api/pith-number/PAEESUG733HZB4UVHFZ3RVSO32/graph.json","fetch_events":"https://pith.science/api/pith-number/PAEESUG733HZB4UVHFZ3RVSO32/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PAEESUG733HZB4UVHFZ3RVSO32/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PAEESUG733HZB4UVHFZ3RVSO32/action/storage_attestation","attest_author":"https://pith.science/pith/PAEESUG733HZB4UVHFZ3RVSO32/action/author_attestation","sign_citation":"https://pith.science/pith/PAEESUG733HZB4UVHFZ3RVSO32/action/citation_signature","submit_replication":"https://pith.science/pith/PAEESUG733HZB4UVHFZ3RVSO32/action/replication_record"}},"created_at":"2026-05-21T01:04:25.830922+00:00","updated_at":"2026-05-21T01:04:25.830922+00:00"}