{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ASVQFVXXNWRKO3SDSNHQX7N35C","short_pith_number":"pith:ASVQFVXX","schema_version":"1.0","canonical_sha256":"04ab02d6f76da2a76e43934f0bfdbbe893f4a8166e2ad39761cb816040d015cf","source":{"kind":"arxiv","id":"2501.14094","version":1},"attestation_state":"computed","paper":{"title":"Datasheets for AI and medical datasets (DAIMS): a data validation and documentation framework before machine learning analysis in medical research","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Anne Svane Frahm (1), Maja Milojevic (1), Ramtin Zargari Marandi (1)","submitted_at":"2025-01-23T21:02:56Z","abstract_excerpt":"Despite progresses in data engineering, there are areas with limited consistencies across data validation and documentation procedures causing confusions and technical problems in research involving machine learning. There have been progresses by introducing frameworks like \"Datasheets for Datasets\", however there are areas for improvements to prepare datasets, ready for ML pipelines. Here, we extend the framework to \"Datasheets for AI and medical datasets - DAIMS.\" Our publicly available solution, DAIMS, provides a checklist including data standardization requirements, a software tool to assi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.14094","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-23T21:02:56Z","cross_cats_sorted":[],"title_canon_sha256":"0e5fc21e82ce9872f6c78c24a8ea23ddb0ee51d5d911a1de056c690fdc80c6f4","abstract_canon_sha256":"f581eaaea8a04071c55d0df37f49303d13331dec513f246ed3d67a9b7093d98e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:04:45.319239Z","signature_b64":"UuG92ZCI7aPGs45FP+e2oSNKB1fXN5UjhLhYGl6I9kOIFvABqX6sFdfdEPNKQ04413UpMgm0qle/Edvf1n9CAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"04ab02d6f76da2a76e43934f0bfdbbe893f4a8166e2ad39761cb816040d015cf","last_reissued_at":"2026-07-05T10:04:45.318856Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:04:45.318856Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Datasheets for AI and medical datasets (DAIMS): a data validation and documentation framework before machine learning analysis in medical research","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Anne Svane Frahm (1), Maja Milojevic (1), Ramtin Zargari Marandi (1)","submitted_at":"2025-01-23T21:02:56Z","abstract_excerpt":"Despite progresses in data engineering, there are areas with limited consistencies across data validation and documentation procedures causing confusions and technical problems in research involving machine learning. There have been progresses by introducing frameworks like \"Datasheets for Datasets\", however there are areas for improvements to prepare datasets, ready for ML pipelines. Here, we extend the framework to \"Datasheets for AI and medical datasets - DAIMS.\" Our publicly available solution, DAIMS, provides a checklist including data standardization requirements, a software tool to assi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.14094","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.14094/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.14094","created_at":"2026-07-05T10:04:45.318915+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.14094v1","created_at":"2026-07-05T10:04:45.318915+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.14094","created_at":"2026-07-05T10:04:45.318915+00:00"},{"alias_kind":"pith_short_12","alias_value":"ASVQFVXXNWRK","created_at":"2026-07-05T10:04:45.318915+00:00"},{"alias_kind":"pith_short_16","alias_value":"ASVQFVXXNWRKO3SD","created_at":"2026-07-05T10:04:45.318915+00:00"},{"alias_kind":"pith_short_8","alias_value":"ASVQFVXX","created_at":"2026-07-05T10:04:45.318915+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ASVQFVXXNWRKO3SDSNHQX7N35C","json":"https://pith.science/pith/ASVQFVXXNWRKO3SDSNHQX7N35C.json","graph_json":"https://pith.science/api/pith-number/ASVQFVXXNWRKO3SDSNHQX7N35C/graph.json","events_json":"https://pith.science/api/pith-number/ASVQFVXXNWRKO3SDSNHQX7N35C/events.json","paper":"https://pith.science/paper/ASVQFVXX"},"agent_actions":{"view_html":"https://pith.science/pith/ASVQFVXXNWRKO3SDSNHQX7N35C","download_json":"https://pith.science/pith/ASVQFVXXNWRKO3SDSNHQX7N35C.json","view_paper":"https://pith.science/paper/ASVQFVXX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.14094&json=true","fetch_graph":"https://pith.science/api/pith-number/ASVQFVXXNWRKO3SDSNHQX7N35C/graph.json","fetch_events":"https://pith.science/api/pith-number/ASVQFVXXNWRKO3SDSNHQX7N35C/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ASVQFVXXNWRKO3SDSNHQX7N35C/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ASVQFVXXNWRKO3SDSNHQX7N35C/action/storage_attestation","attest_author":"https://pith.science/pith/ASVQFVXXNWRKO3SDSNHQX7N35C/action/author_attestation","sign_citation":"https://pith.science/pith/ASVQFVXXNWRKO3SDSNHQX7N35C/action/citation_signature","submit_replication":"https://pith.science/pith/ASVQFVXXNWRKO3SDSNHQX7N35C/action/replication_record"}},"created_at":"2026-07-05T10:04:45.318915+00:00","updated_at":"2026-07-05T10:04:45.318915+00:00"}