{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:O26F3NB2CHU3YVKKT52SWZTIH5","short_pith_number":"pith:O26F3NB2","schema_version":"1.0","canonical_sha256":"76bc5db43a11e9bc554a9f752b66683f54c8f5ada99b7851e449c512fd6dbab0","source":{"kind":"arxiv","id":"2407.09293","version":2},"attestation_state":"computed","paper":{"title":"A decomposition of Fisher's information to inform sample size for developing fair and precise clinical prediction models -- part 1: binary outcomes","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"stat.ME","authors_text":"Alastair Denniston, Amardeep Legha, Frank E Harrell Jr, Gary S Collins, Glen P Martin, Joie Ensor, Kym IE Snell, Laura Kirton, Laure Wynants, Lucinda Archer, Paula Dhiman, Rebecca Whittle, Richard D Riley, Xiaoxuan Liu","submitted_at":"2024-07-12T14:27:11Z","abstract_excerpt":"When developing a clinical prediction model, the sample size of the development dataset is a key consideration. Small sample sizes lead to greater concerns of overfitting, instability, poor performance and lack of fairness. Previous research has outlined minimum sample size calculations to minimise overfitting and precisely estimate the overall risk. However even when meeting these criteria, the uncertainty (instability) in individual-level risk estimates may be considerable. In this article we propose how to examine and calculate the sample size required for developing a model with acceptably"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.09293","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"stat.ME","submitted_at":"2024-07-12T14:27:11Z","cross_cats_sorted":[],"title_canon_sha256":"2bb07f030cb27eb3d78f498cdc9647f83cf88c8e1fcfed30903c83197a689126","abstract_canon_sha256":"b9c117b549c58b94e7dde717a5f58ffea9684b6311509aec3fc7233900e90ac6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:04:57.460273Z","signature_b64":"udNyfL1JhPPAn9TE4+iTA9ES3p/icAiV4QSpfM2/likExGLu8vWVO7QnKgrNMzTCV/yeh3Z6IQfkNHipMyg3DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"76bc5db43a11e9bc554a9f752b66683f54c8f5ada99b7851e449c512fd6dbab0","last_reissued_at":"2026-07-05T10:04:57.459837Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:04:57.459837Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A decomposition of Fisher's information to inform sample size for developing fair and precise clinical prediction models -- part 1: binary outcomes","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"stat.ME","authors_text":"Alastair Denniston, Amardeep Legha, Frank E Harrell Jr, Gary S Collins, Glen P Martin, Joie Ensor, Kym IE Snell, Laura Kirton, Laure Wynants, Lucinda Archer, Paula Dhiman, Rebecca Whittle, Richard D Riley, Xiaoxuan Liu","submitted_at":"2024-07-12T14:27:11Z","abstract_excerpt":"When developing a clinical prediction model, the sample size of the development dataset is a key consideration. Small sample sizes lead to greater concerns of overfitting, instability, poor performance and lack of fairness. Previous research has outlined minimum sample size calculations to minimise overfitting and precisely estimate the overall risk. However even when meeting these criteria, the uncertainty (instability) in individual-level risk estimates may be considerable. In this article we propose how to examine and calculate the sample size required for developing a model with acceptably"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.09293","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.09293/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.09293","created_at":"2026-07-05T10:04:57.459891+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.09293v2","created_at":"2026-07-05T10:04:57.459891+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.09293","created_at":"2026-07-05T10:04:57.459891+00:00"},{"alias_kind":"pith_short_12","alias_value":"O26F3NB2CHU3","created_at":"2026-07-05T10:04:57.459891+00:00"},{"alias_kind":"pith_short_16","alias_value":"O26F3NB2CHU3YVKK","created_at":"2026-07-05T10:04:57.459891+00:00"},{"alias_kind":"pith_short_8","alias_value":"O26F3NB2","created_at":"2026-07-05T10:04:57.459891+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.23548","citing_title":"A decomposition of Fisher's information to inform sample size for developing or updating fair and precise clinical prediction models -- Part 3: continuous outcomes","ref_index":14,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O26F3NB2CHU3YVKKT52SWZTIH5","json":"https://pith.science/pith/O26F3NB2CHU3YVKKT52SWZTIH5.json","graph_json":"https://pith.science/api/pith-number/O26F3NB2CHU3YVKKT52SWZTIH5/graph.json","events_json":"https://pith.science/api/pith-number/O26F3NB2CHU3YVKKT52SWZTIH5/events.json","paper":"https://pith.science/paper/O26F3NB2"},"agent_actions":{"view_html":"https://pith.science/pith/O26F3NB2CHU3YVKKT52SWZTIH5","download_json":"https://pith.science/pith/O26F3NB2CHU3YVKKT52SWZTIH5.json","view_paper":"https://pith.science/paper/O26F3NB2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.09293&json=true","fetch_graph":"https://pith.science/api/pith-number/O26F3NB2CHU3YVKKT52SWZTIH5/graph.json","fetch_events":"https://pith.science/api/pith-number/O26F3NB2CHU3YVKKT52SWZTIH5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O26F3NB2CHU3YVKKT52SWZTIH5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O26F3NB2CHU3YVKKT52SWZTIH5/action/storage_attestation","attest_author":"https://pith.science/pith/O26F3NB2CHU3YVKKT52SWZTIH5/action/author_attestation","sign_citation":"https://pith.science/pith/O26F3NB2CHU3YVKKT52SWZTIH5/action/citation_signature","submit_replication":"https://pith.science/pith/O26F3NB2CHU3YVKKT52SWZTIH5/action/replication_record"}},"created_at":"2026-07-05T10:04:57.459891+00:00","updated_at":"2026-07-05T10:04:57.459891+00:00"}