{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CQP5RSMLWJ4636U5Z5FQ6Q37RO","short_pith_number":"pith:CQP5RSML","schema_version":"1.0","canonical_sha256":"141fd8c98bb279edfa9dcf4b0f437f8ba1c8f3ae702c07aea7006ac1dc47ab8c","source":{"kind":"arxiv","id":"2512.22240","version":5},"attestation_state":"computed","paper":{"title":"EvoXplain: When Machine Learning Models Agree on Predictions but Disagree on Why -- Measuring Mechanistic Multiplicity Across Training Runs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Chama Bensmail","submitted_at":"2025-12-23T18:34:51Z","abstract_excerpt":"Machine learning models are primarily judged by predictive performance, especially in applied genomics, where explanations are read as biological findings. In practice, reported gene panels are stabilised by averaging, ranking, or taking consensus over the many models a pipeline produces across cross-validation folds, tuning grids, and repeated runs. This raises an overlooked question: when two models achieve high accuracy, do they rely on the same internal logic, or reach the same outcome via different mechanisms? We introduce EvoXplain, a diagnostic framework that measures whether a pipeline"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2512.22240","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-12-23T18:34:51Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"9f548db225a409fd9180fe44434d5afd79b8595cd77998bcdcc3a2018d532be7","abstract_canon_sha256":"d46a0596cc364c997a0e496b6161519d40da03d51cec4efb6402c37080b53709"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-07T02:19:44.239475Z","signature_b64":"BqP099q9QPgRhp4WBJ/tHMujhIiafuyGONr9KBYEJ6WK1ZUFExni4IfqfwlfZ3FSR9Q2cbYtf7ApTX7+gE77CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"141fd8c98bb279edfa9dcf4b0f437f8ba1c8f3ae702c07aea7006ac1dc47ab8c","last_reissued_at":"2026-07-07T02:19:44.238633Z","signature_status":"signed_v1","first_computed_at":"2026-07-07T02:19:44.238633Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EvoXplain: When Machine Learning Models Agree on Predictions but Disagree on Why -- Measuring Mechanistic Multiplicity Across Training Runs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Chama Bensmail","submitted_at":"2025-12-23T18:34:51Z","abstract_excerpt":"Machine learning models are primarily judged by predictive performance, especially in applied genomics, where explanations are read as biological findings. In practice, reported gene panels are stabilised by averaging, ranking, or taking consensus over the many models a pipeline produces across cross-validation folds, tuning grids, and repeated runs. This raises an overlooked question: when two models achieve high accuracy, do they rely on the same internal logic, or reach the same outcome via different mechanisms? We introduce EvoXplain, a diagnostic framework that measures whether a pipeline"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2512.22240","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2512.22240/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2512.22240","created_at":"2026-07-07T02:19:44.238735+00:00"},{"alias_kind":"arxiv_version","alias_value":"2512.22240v5","created_at":"2026-07-07T02:19:44.238735+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2512.22240","created_at":"2026-07-07T02:19:44.238735+00:00"},{"alias_kind":"pith_short_12","alias_value":"CQP5RSMLWJ46","created_at":"2026-07-07T02:19:44.238735+00:00"},{"alias_kind":"pith_short_16","alias_value":"CQP5RSMLWJ4636U5","created_at":"2026-07-07T02:19:44.238735+00:00"},{"alias_kind":"pith_short_8","alias_value":"CQP5RSML","created_at":"2026-07-07T02:19:44.238735+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CQP5RSMLWJ4636U5Z5FQ6Q37RO","json":"https://pith.science/pith/CQP5RSMLWJ4636U5Z5FQ6Q37RO.json","graph_json":"https://pith.science/api/pith-number/CQP5RSMLWJ4636U5Z5FQ6Q37RO/graph.json","events_json":"https://pith.science/api/pith-number/CQP5RSMLWJ4636U5Z5FQ6Q37RO/events.json","paper":"https://pith.science/paper/CQP5RSML"},"agent_actions":{"view_html":"https://pith.science/pith/CQP5RSMLWJ4636U5Z5FQ6Q37RO","download_json":"https://pith.science/pith/CQP5RSMLWJ4636U5Z5FQ6Q37RO.json","view_paper":"https://pith.science/paper/CQP5RSML","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2512.22240&json=true","fetch_graph":"https://pith.science/api/pith-number/CQP5RSMLWJ4636U5Z5FQ6Q37RO/graph.json","fetch_events":"https://pith.science/api/pith-number/CQP5RSMLWJ4636U5Z5FQ6Q37RO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CQP5RSMLWJ4636U5Z5FQ6Q37RO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CQP5RSMLWJ4636U5Z5FQ6Q37RO/action/storage_attestation","attest_author":"https://pith.science/pith/CQP5RSMLWJ4636U5Z5FQ6Q37RO/action/author_attestation","sign_citation":"https://pith.science/pith/CQP5RSMLWJ4636U5Z5FQ6Q37RO/action/citation_signature","submit_replication":"https://pith.science/pith/CQP5RSMLWJ4636U5Z5FQ6Q37RO/action/replication_record"}},"created_at":"2026-07-07T02:19:44.238735+00:00","updated_at":"2026-07-07T02:19:44.238735+00:00"}