{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:QABSZRFHKHTGL2JEFPIOUZ7ACZ","short_pith_number":"pith:QABSZRFH","schema_version":"1.0","canonical_sha256":"80032cc4a751e665e9242bd0ea67e01662a2873df944fad4cc7b737c1a8532f5","source":{"kind":"arxiv","id":"2010.15100","version":2},"attestation_state":"computed","paper":{"title":"Evaluating Model Robustness and Stability to Dataset Shift","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Adarsh Subbaswamy, Roy Adams, Suchi Saria","submitted_at":"2020-10-28T17:35:39Z","abstract_excerpt":"As the use of machine learning in high impact domains becomes widespread, the importance of evaluating safety has increased. An important aspect of this is evaluating how robust a model is to changes in setting or population, which typically requires applying the model to multiple, independent datasets. Since the cost of collecting such datasets is often prohibitive, in this paper, we propose a framework for analyzing this type of stability using the available data. We use the original evaluation data to determine distributions under which the algorithm performs poorly, and estimate the algori"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2010.15100","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-10-28T17:35:39Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"29c0358d21d27d71a8d37f58edff9f118699276092ce1698fa1c3db14321d51f","abstract_canon_sha256":"07b7a09de84e751e8dd53e836c9a4e4353d35aee752ef966c4a62684da161816"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:22:57.309414Z","signature_b64":"KVVAGxn0ZaFs/F7W6zMzgxQ30WXum5DReSvmMSuQacuH/Tr6HBCYR89cogwXX9MAVlJpHTcP9dMNMSfxChuWAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"80032cc4a751e665e9242bd0ea67e01662a2873df944fad4cc7b737c1a8532f5","last_reissued_at":"2026-07-05T02:22:57.309022Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:22:57.309022Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating Model Robustness and Stability to Dataset Shift","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Adarsh Subbaswamy, Roy Adams, Suchi Saria","submitted_at":"2020-10-28T17:35:39Z","abstract_excerpt":"As the use of machine learning in high impact domains becomes widespread, the importance of evaluating safety has increased. An important aspect of this is evaluating how robust a model is to changes in setting or population, which typically requires applying the model to multiple, independent datasets. Since the cost of collecting such datasets is often prohibitive, in this paper, we propose a framework for analyzing this type of stability using the available data. We use the original evaluation data to determine distributions under which the algorithm performs poorly, and estimate the algori"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2010.15100","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2010.15100/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2010.15100","created_at":"2026-07-05T02:22:57.309071+00:00"},{"alias_kind":"arxiv_version","alias_value":"2010.15100v2","created_at":"2026-07-05T02:22:57.309071+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2010.15100","created_at":"2026-07-05T02:22:57.309071+00:00"},{"alias_kind":"pith_short_12","alias_value":"QABSZRFHKHTG","created_at":"2026-07-05T02:22:57.309071+00:00"},{"alias_kind":"pith_short_16","alias_value":"QABSZRFHKHTGL2JE","created_at":"2026-07-05T02:22:57.309071+00:00"},{"alias_kind":"pith_short_8","alias_value":"QABSZRFH","created_at":"2026-07-05T02:22:57.309071+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.04275","citing_title":"Scoping review of methodology for aiding generalisability and transportability of clinical prediction models","ref_index":45,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QABSZRFHKHTGL2JEFPIOUZ7ACZ","json":"https://pith.science/pith/QABSZRFHKHTGL2JEFPIOUZ7ACZ.json","graph_json":"https://pith.science/api/pith-number/QABSZRFHKHTGL2JEFPIOUZ7ACZ/graph.json","events_json":"https://pith.science/api/pith-number/QABSZRFHKHTGL2JEFPIOUZ7ACZ/events.json","paper":"https://pith.science/paper/QABSZRFH"},"agent_actions":{"view_html":"https://pith.science/pith/QABSZRFHKHTGL2JEFPIOUZ7ACZ","download_json":"https://pith.science/pith/QABSZRFHKHTGL2JEFPIOUZ7ACZ.json","view_paper":"https://pith.science/paper/QABSZRFH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2010.15100&json=true","fetch_graph":"https://pith.science/api/pith-number/QABSZRFHKHTGL2JEFPIOUZ7ACZ/graph.json","fetch_events":"https://pith.science/api/pith-number/QABSZRFHKHTGL2JEFPIOUZ7ACZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QABSZRFHKHTGL2JEFPIOUZ7ACZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QABSZRFHKHTGL2JEFPIOUZ7ACZ/action/storage_attestation","attest_author":"https://pith.science/pith/QABSZRFHKHTGL2JEFPIOUZ7ACZ/action/author_attestation","sign_citation":"https://pith.science/pith/QABSZRFHKHTGL2JEFPIOUZ7ACZ/action/citation_signature","submit_replication":"https://pith.science/pith/QABSZRFHKHTGL2JEFPIOUZ7ACZ/action/replication_record"}},"created_at":"2026-07-05T02:22:57.309071+00:00","updated_at":"2026-07-05T02:22:57.309071+00:00"}