{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:HMUPI2EQ2O4WF4DZ4V6FY2AERY","short_pith_number":"pith:HMUPI2EQ","schema_version":"1.0","canonical_sha256":"3b28f46890d3b962f079e57c5c68048e3e5a214f4d5a1a786a4b3a9d68d9f235","source":{"kind":"arxiv","id":"2502.04397","version":3},"attestation_state":"computed","paper":{"title":"Multimodal Medical Code Tokenizer","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Faryad Sahneh, Lukas Fesser, Marinka Zitnik, Ruth Johnson, Shanghua Gao, Shvat Messica, Xiaorui Su, Yepeng Huang","submitted_at":"2025-02-06T06:58:09Z","abstract_excerpt":"Foundation models trained on patient electronic health records (EHRs) require tokenizing medical data into sequences of discrete vocabulary items. Existing tokenizers treat medical codes from EHRs as isolated textual tokens. However, each medical code is defined by its textual description, its position in ontological hierarchies, and its relationships to other codes, such as disease co-occurrences and drug-treatment associations. Medical vocabularies contain more than 600,000 codes with critical information for clinical reasoning. We introduce MedTok, a multimodal medical code tokenizer that u"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.04397","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-06T06:58:09Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"f59691f6010e2d33c577e7a37a6aad3d4776709ea739b6f3d7bb7d7419865bdb","abstract_canon_sha256":"09d055008647b1d8d7431781b9b52ca7b272f76cbb55f9c9b45aa8208c884bda"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:28:33.817633Z","signature_b64":"do3kpmlnu5nrUscURWmAngl33JvuPRu3bLGVS4hyMNh3F2Y/Tpjik7gUdenPKN8EcbJJ3hpkS9WG9lvkpEMkBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3b28f46890d3b962f079e57c5c68048e3e5a214f4d5a1a786a4b3a9d68d9f235","last_reissued_at":"2026-07-05T11:28:33.817127Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:28:33.817127Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multimodal Medical Code Tokenizer","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Faryad Sahneh, Lukas Fesser, Marinka Zitnik, Ruth Johnson, Shanghua Gao, Shvat Messica, Xiaorui Su, Yepeng Huang","submitted_at":"2025-02-06T06:58:09Z","abstract_excerpt":"Foundation models trained on patient electronic health records (EHRs) require tokenizing medical data into sequences of discrete vocabulary items. Existing tokenizers treat medical codes from EHRs as isolated textual tokens. However, each medical code is defined by its textual description, its position in ontological hierarchies, and its relationships to other codes, such as disease co-occurrences and drug-treatment associations. Medical vocabularies contain more than 600,000 codes with critical information for clinical reasoning. We introduce MedTok, a multimodal medical code tokenizer that u"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.04397","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.04397/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.04397","created_at":"2026-07-05T11:28:33.817188+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.04397v3","created_at":"2026-07-05T11:28:33.817188+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.04397","created_at":"2026-07-05T11:28:33.817188+00:00"},{"alias_kind":"pith_short_12","alias_value":"HMUPI2EQ2O4W","created_at":"2026-07-05T11:28:33.817188+00:00"},{"alias_kind":"pith_short_16","alias_value":"HMUPI2EQ2O4WF4DZ","created_at":"2026-07-05T11:28:33.817188+00:00"},{"alias_kind":"pith_short_8","alias_value":"HMUPI2EQ","created_at":"2026-07-05T11:28:33.817188+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.16775","citing_title":"Representation Before Training: A Fixed-Budget Benchmark for Generative Medical Event Models","ref_index":37,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HMUPI2EQ2O4WF4DZ4V6FY2AERY","json":"https://pith.science/pith/HMUPI2EQ2O4WF4DZ4V6FY2AERY.json","graph_json":"https://pith.science/api/pith-number/HMUPI2EQ2O4WF4DZ4V6FY2AERY/graph.json","events_json":"https://pith.science/api/pith-number/HMUPI2EQ2O4WF4DZ4V6FY2AERY/events.json","paper":"https://pith.science/paper/HMUPI2EQ"},"agent_actions":{"view_html":"https://pith.science/pith/HMUPI2EQ2O4WF4DZ4V6FY2AERY","download_json":"https://pith.science/pith/HMUPI2EQ2O4WF4DZ4V6FY2AERY.json","view_paper":"https://pith.science/paper/HMUPI2EQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.04397&json=true","fetch_graph":"https://pith.science/api/pith-number/HMUPI2EQ2O4WF4DZ4V6FY2AERY/graph.json","fetch_events":"https://pith.science/api/pith-number/HMUPI2EQ2O4WF4DZ4V6FY2AERY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HMUPI2EQ2O4WF4DZ4V6FY2AERY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HMUPI2EQ2O4WF4DZ4V6FY2AERY/action/storage_attestation","attest_author":"https://pith.science/pith/HMUPI2EQ2O4WF4DZ4V6FY2AERY/action/author_attestation","sign_citation":"https://pith.science/pith/HMUPI2EQ2O4WF4DZ4V6FY2AERY/action/citation_signature","submit_replication":"https://pith.science/pith/HMUPI2EQ2O4WF4DZ4V6FY2AERY/action/replication_record"}},"created_at":"2026-07-05T11:28:33.817188+00:00","updated_at":"2026-07-05T11:28:33.817188+00:00"}