{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DIGLZUVKEK2V42FKUOPK6BEL3Y","short_pith_number":"pith:DIGLZUVK","schema_version":"1.0","canonical_sha256":"1a0cbcd2aa22b55e68aaa39eaf048bde0f2e0fb44eff0bcb9e404ac59753b9b3","source":{"kind":"arxiv","id":"2506.03913","version":1},"attestation_state":"computed","paper":{"title":"When Fairness Isn't Statistical: The Limits of Machine Learning in Evaluating Legal Reasoning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Claire Barale, Michael Rovatsos, Nehal Bhuta","submitted_at":"2025-06-04T13:05:37Z","abstract_excerpt":"Legal decisions are increasingly evaluated for fairness, consistency, and bias using machine learning (ML) techniques. In high-stakes domains like refugee adjudication, such methods are often applied to detect disparities in outcomes. Yet it remains unclear whether statistical methods can meaningfully assess fairness in legal contexts shaped by discretion, normative complexity, and limited ground truth.\n  In this paper, we empirically evaluate three common ML approaches (feature-based analysis, semantic clustering, and predictive modeling) on a large, real-world dataset of 59,000+ Canadian ref"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.03913","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-04T13:05:37Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"1007926aeaa0d3fd1f7d20f6fbfef629914a5dbe9f4892570e768582f6f1ecc5","abstract_canon_sha256":"2be9d37728340bd91fa3c65225eb5e8acf87f0b3a786984d7c2db681903296eb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:15:53.974607Z","signature_b64":"padvYStdGApm9prKh5VKtIOCDXmK5eDNSiVNMFkaYFyaENZ1NcNbbbVSQBRWLDbTHtREe5phQQS8AH5J9+wNBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1a0cbcd2aa22b55e68aaa39eaf048bde0f2e0fb44eff0bcb9e404ac59753b9b3","last_reissued_at":"2026-07-05T11:15:53.974126Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:15:53.974126Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"When Fairness Isn't Statistical: The Limits of Machine Learning in Evaluating Legal Reasoning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Claire Barale, Michael Rovatsos, Nehal Bhuta","submitted_at":"2025-06-04T13:05:37Z","abstract_excerpt":"Legal decisions are increasingly evaluated for fairness, consistency, and bias using machine learning (ML) techniques. In high-stakes domains like refugee adjudication, such methods are often applied to detect disparities in outcomes. Yet it remains unclear whether statistical methods can meaningfully assess fairness in legal contexts shaped by discretion, normative complexity, and limited ground truth.\n  In this paper, we empirically evaluate three common ML approaches (feature-based analysis, semantic clustering, and predictive modeling) on a large, real-world dataset of 59,000+ Canadian ref"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.03913","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.03913/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.03913","created_at":"2026-07-05T11:15:53.974201+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.03913v1","created_at":"2026-07-05T11:15:53.974201+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.03913","created_at":"2026-07-05T11:15:53.974201+00:00"},{"alias_kind":"pith_short_12","alias_value":"DIGLZUVKEK2V","created_at":"2026-07-05T11:15:53.974201+00:00"},{"alias_kind":"pith_short_16","alias_value":"DIGLZUVKEK2V42FK","created_at":"2026-07-05T11:15:53.974201+00:00"},{"alias_kind":"pith_short_8","alias_value":"DIGLZUVK","created_at":"2026-07-05T11:15:53.974201+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09809","citing_title":"Evaluation Cards: An Interpretive Layer for AI Evaluation Reporting","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DIGLZUVKEK2V42FKUOPK6BEL3Y","json":"https://pith.science/pith/DIGLZUVKEK2V42FKUOPK6BEL3Y.json","graph_json":"https://pith.science/api/pith-number/DIGLZUVKEK2V42FKUOPK6BEL3Y/graph.json","events_json":"https://pith.science/api/pith-number/DIGLZUVKEK2V42FKUOPK6BEL3Y/events.json","paper":"https://pith.science/paper/DIGLZUVK"},"agent_actions":{"view_html":"https://pith.science/pith/DIGLZUVKEK2V42FKUOPK6BEL3Y","download_json":"https://pith.science/pith/DIGLZUVKEK2V42FKUOPK6BEL3Y.json","view_paper":"https://pith.science/paper/DIGLZUVK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.03913&json=true","fetch_graph":"https://pith.science/api/pith-number/DIGLZUVKEK2V42FKUOPK6BEL3Y/graph.json","fetch_events":"https://pith.science/api/pith-number/DIGLZUVKEK2V42FKUOPK6BEL3Y/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DIGLZUVKEK2V42FKUOPK6BEL3Y/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DIGLZUVKEK2V42FKUOPK6BEL3Y/action/storage_attestation","attest_author":"https://pith.science/pith/DIGLZUVKEK2V42FKUOPK6BEL3Y/action/author_attestation","sign_citation":"https://pith.science/pith/DIGLZUVKEK2V42FKUOPK6BEL3Y/action/citation_signature","submit_replication":"https://pith.science/pith/DIGLZUVKEK2V42FKUOPK6BEL3Y/action/replication_record"}},"created_at":"2026-07-05T11:15:53.974201+00:00","updated_at":"2026-07-05T11:15:53.974201+00:00"}