{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:7PBF2UPRBFPEC3IEFUXKFTLUB2","short_pith_number":"pith:7PBF2UPR","schema_version":"1.0","canonical_sha256":"fbc25d51f1095e416d042d2ea2cd740eb98636c6a01798f8add38bd65f9f7b40","source":{"kind":"arxiv","id":"2503.10837","version":2},"attestation_state":"computed","paper":{"title":"Lessons from the trenches on evaluating machine-learning systems in materials science","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cond-mat.mtrl-sci","authors_text":"Kevin Maik Jablonka, Mara Schilling-Wilhelmi, Nawaf Alampara","submitted_at":"2025-03-13T19:40:58Z","abstract_excerpt":"Measurements are fundamental to knowledge creation in science, enabling consistent sharing of findings and serving as the foundation for scientific discovery. As machine learning systems increasingly transform scientific fields, the question of how to effectively evaluate these systems becomes crucial for ensuring reliable progress.\n  In this review, we examine the current state and future directions of evaluation frameworks for machine learning in science. We organize the review around a broadly applicable framework for evaluating machine learning systems through the lens of statistical measu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.10837","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cond-mat.mtrl-sci","submitted_at":"2025-03-13T19:40:58Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"5b41244b0fdcb1345d46bf0d38adf823e8584fb4fad0329b323f0294479cdfcb","abstract_canon_sha256":"60d90f445a0dcd606de6580ffbbbae9ed1f2ac67421545b450149f484e690668"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:59:07.316310Z","signature_b64":"eXjIFZcw3C8dHqJnoKM1dN1I+C3yqf5mttgpEIWTue7PUsZ1+nSd+KUGG51eledxTI9YExaShxQrWqpsecPNCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fbc25d51f1095e416d042d2ea2cd740eb98636c6a01798f8add38bd65f9f7b40","last_reissued_at":"2026-07-05T10:59:07.315801Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:59:07.315801Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Lessons from the trenches on evaluating machine-learning systems in materials science","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cond-mat.mtrl-sci","authors_text":"Kevin Maik Jablonka, Mara Schilling-Wilhelmi, Nawaf Alampara","submitted_at":"2025-03-13T19:40:58Z","abstract_excerpt":"Measurements are fundamental to knowledge creation in science, enabling consistent sharing of findings and serving as the foundation for scientific discovery. As machine learning systems increasingly transform scientific fields, the question of how to effectively evaluate these systems becomes crucial for ensuring reliable progress.\n  In this review, we examine the current state and future directions of evaluation frameworks for machine learning in science. We organize the review around a broadly applicable framework for evaluating machine learning systems through the lens of statistical measu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.10837","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.10837/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.10837","created_at":"2026-07-05T10:59:07.315865+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.10837v2","created_at":"2026-07-05T10:59:07.315865+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.10837","created_at":"2026-07-05T10:59:07.315865+00:00"},{"alias_kind":"pith_short_12","alias_value":"7PBF2UPRBFPE","created_at":"2026-07-05T10:59:07.315865+00:00"},{"alias_kind":"pith_short_16","alias_value":"7PBF2UPRBFPEC3IE","created_at":"2026-07-05T10:59:07.315865+00:00"},{"alias_kind":"pith_short_8","alias_value":"7PBF2UPR","created_at":"2026-07-05T10:59:07.315865+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.18086","citing_title":"Materials Informatics Across the Length Scales","ref_index":157,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7PBF2UPRBFPEC3IEFUXKFTLUB2","json":"https://pith.science/pith/7PBF2UPRBFPEC3IEFUXKFTLUB2.json","graph_json":"https://pith.science/api/pith-number/7PBF2UPRBFPEC3IEFUXKFTLUB2/graph.json","events_json":"https://pith.science/api/pith-number/7PBF2UPRBFPEC3IEFUXKFTLUB2/events.json","paper":"https://pith.science/paper/7PBF2UPR"},"agent_actions":{"view_html":"https://pith.science/pith/7PBF2UPRBFPEC3IEFUXKFTLUB2","download_json":"https://pith.science/pith/7PBF2UPRBFPEC3IEFUXKFTLUB2.json","view_paper":"https://pith.science/paper/7PBF2UPR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.10837&json=true","fetch_graph":"https://pith.science/api/pith-number/7PBF2UPRBFPEC3IEFUXKFTLUB2/graph.json","fetch_events":"https://pith.science/api/pith-number/7PBF2UPRBFPEC3IEFUXKFTLUB2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7PBF2UPRBFPEC3IEFUXKFTLUB2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7PBF2UPRBFPEC3IEFUXKFTLUB2/action/storage_attestation","attest_author":"https://pith.science/pith/7PBF2UPRBFPEC3IEFUXKFTLUB2/action/author_attestation","sign_citation":"https://pith.science/pith/7PBF2UPRBFPEC3IEFUXKFTLUB2/action/citation_signature","submit_replication":"https://pith.science/pith/7PBF2UPRBFPEC3IEFUXKFTLUB2/action/replication_record"}},"created_at":"2026-07-05T10:59:07.315865+00:00","updated_at":"2026-07-05T10:59:07.315865+00:00"}