{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:24SCSPPC7O5225LYH6APHLE7GL","short_pith_number":"pith:24SCSPPC","schema_version":"1.0","canonical_sha256":"d724293de2fbbbad75783f80f3ac9f32df601ed46c183a7786929096220d2d88","source":{"kind":"arxiv","id":"2505.22356","version":1},"attestation_state":"computed","paper":{"title":"Suitability Filter: A Statistical Framework for Classifier Evaluation in Real-World Deployment Settings","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CY","stat.ML"],"primary_cat":"cs.LG","authors_text":"Ang\\'eline Pouget, Mohammad Yaghini, Nicolas Papernot, Stephan Rabanser","submitted_at":"2025-05-28T13:37:04Z","abstract_excerpt":"Deploying machine learning models in safety-critical domains poses a key challenge: ensuring reliable model performance on downstream user data without access to ground truth labels for direct validation. We propose the suitability filter, a novel framework designed to detect performance deterioration by utilizing suitability signals -- model output features that are sensitive to covariate shifts and indicative of potential prediction errors. The suitability filter evaluates whether classifier accuracy on unlabeled user data shows significant degradation compared to the accuracy measured on th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.22356","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-28T13:37:04Z","cross_cats_sorted":["cs.AI","cs.CY","stat.ML"],"title_canon_sha256":"68c48ecc26e83bf4307369a43886a616ac5bc6e151ef2e6aa2ea59f5c392fe9f","abstract_canon_sha256":"a8bf2bb69a2fea4d2ac5f771e8e726eb166477b67844c23d84b8445f11c7f4e7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:11:13.825710Z","signature_b64":"3zOgQFsDcgOx2RyWJ28ZFfdGu82oUtMJVRXFKQUy3wmpRHcsZW9S6LQCTvORZEmX5ukdB+87+HJ5Bi6E/bL6DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d724293de2fbbbad75783f80f3ac9f32df601ed46c183a7786929096220d2d88","last_reissued_at":"2026-07-05T11:11:13.825105Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:11:13.825105Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Suitability Filter: A Statistical Framework for Classifier Evaluation in Real-World Deployment Settings","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CY","stat.ML"],"primary_cat":"cs.LG","authors_text":"Ang\\'eline Pouget, Mohammad Yaghini, Nicolas Papernot, Stephan Rabanser","submitted_at":"2025-05-28T13:37:04Z","abstract_excerpt":"Deploying machine learning models in safety-critical domains poses a key challenge: ensuring reliable model performance on downstream user data without access to ground truth labels for direct validation. We propose the suitability filter, a novel framework designed to detect performance deterioration by utilizing suitability signals -- model output features that are sensitive to covariate shifts and indicative of potential prediction errors. The suitability filter evaluates whether classifier accuracy on unlabeled user data shows significant degradation compared to the accuracy measured on th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.22356","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.22356/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.22356","created_at":"2026-07-05T11:11:13.825169+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.22356v1","created_at":"2026-07-05T11:11:13.825169+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.22356","created_at":"2026-07-05T11:11:13.825169+00:00"},{"alias_kind":"pith_short_12","alias_value":"24SCSPPC7O52","created_at":"2026-07-05T11:11:13.825169+00:00"},{"alias_kind":"pith_short_16","alias_value":"24SCSPPC7O5225LY","created_at":"2026-07-05T11:11:13.825169+00:00"},{"alias_kind":"pith_short_8","alias_value":"24SCSPPC","created_at":"2026-07-05T11:11:13.825169+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19594","citing_title":"Unsupervised Causal Abstractions Discovery","ref_index":57,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/24SCSPPC7O5225LYH6APHLE7GL","json":"https://pith.science/pith/24SCSPPC7O5225LYH6APHLE7GL.json","graph_json":"https://pith.science/api/pith-number/24SCSPPC7O5225LYH6APHLE7GL/graph.json","events_json":"https://pith.science/api/pith-number/24SCSPPC7O5225LYH6APHLE7GL/events.json","paper":"https://pith.science/paper/24SCSPPC"},"agent_actions":{"view_html":"https://pith.science/pith/24SCSPPC7O5225LYH6APHLE7GL","download_json":"https://pith.science/pith/24SCSPPC7O5225LYH6APHLE7GL.json","view_paper":"https://pith.science/paper/24SCSPPC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.22356&json=true","fetch_graph":"https://pith.science/api/pith-number/24SCSPPC7O5225LYH6APHLE7GL/graph.json","fetch_events":"https://pith.science/api/pith-number/24SCSPPC7O5225LYH6APHLE7GL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/24SCSPPC7O5225LYH6APHLE7GL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/24SCSPPC7O5225LYH6APHLE7GL/action/storage_attestation","attest_author":"https://pith.science/pith/24SCSPPC7O5225LYH6APHLE7GL/action/author_attestation","sign_citation":"https://pith.science/pith/24SCSPPC7O5225LYH6APHLE7GL/action/citation_signature","submit_replication":"https://pith.science/pith/24SCSPPC7O5225LYH6APHLE7GL/action/replication_record"}},"created_at":"2026-07-05T11:11:13.825169+00:00","updated_at":"2026-07-05T11:11:13.825169+00:00"}