{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:HHMBAQWPGAEBMBUMVPELHWB76S","short_pith_number":"pith:HHMBAQWP","schema_version":"1.0","canonical_sha256":"39d81042cf300816068cabc8b3d83ff49fdeb853a0386a6eeb5774e2dcbe8869","source":{"kind":"arxiv","id":"2306.05052","version":1},"attestation_state":"computed","paper":{"title":"Interpretable Medical Diagnostics with Structured Data Extraction by Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Aleksa Bisercic, Andrija Petrovic, Boris Delibasic, Mihaela van der Schaar, Mladen Nikolic, Pietro Lio","submitted_at":"2023-06-08T09:12:28Z","abstract_excerpt":"Tabular data is often hidden in text, particularly in medical diagnostic reports. Traditional machine learning (ML) models designed to work with tabular data, cannot effectively process information in such form. On the other hand, large language models (LLMs) which excel at textual tasks, are probably not the best tool for modeling tabular data. Therefore, we propose a novel, simple, and effective methodology for extracting structured tabular data from textual medical reports, called TEMED-LLM. Drawing upon the reasoning capabilities of LLMs, TEMED-LLM goes beyond traditional extraction techni"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.05052","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-06-08T09:12:28Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"2da07aff5db5b6e84a03ea332879d2e7b73ace5b0b95776c297f40890c830a2c","abstract_canon_sha256":"357788cea824b7dae72540f1c5b3cd463005a65fb6e0cb976ca68c9c6cad251e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:18:45.158306Z","signature_b64":"TMVU8V7B59ZRRXVQMRMPGRbxW87gNavkeajmdjIXQfI2wxVPp5PZw/qHUlyPrakD5k7HAa9vaBkbE7gFr9NiAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"39d81042cf300816068cabc8b3d83ff49fdeb853a0386a6eeb5774e2dcbe8869","last_reissued_at":"2026-07-05T06:18:45.157835Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:18:45.157835Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Interpretable Medical Diagnostics with Structured Data Extraction by Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Aleksa Bisercic, Andrija Petrovic, Boris Delibasic, Mihaela van der Schaar, Mladen Nikolic, Pietro Lio","submitted_at":"2023-06-08T09:12:28Z","abstract_excerpt":"Tabular data is often hidden in text, particularly in medical diagnostic reports. Traditional machine learning (ML) models designed to work with tabular data, cannot effectively process information in such form. On the other hand, large language models (LLMs) which excel at textual tasks, are probably not the best tool for modeling tabular data. Therefore, we propose a novel, simple, and effective methodology for extracting structured tabular data from textual medical reports, called TEMED-LLM. Drawing upon the reasoning capabilities of LLMs, TEMED-LLM goes beyond traditional extraction techni"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.05052","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.05052/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.05052","created_at":"2026-07-05T06:18:45.157892+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.05052v1","created_at":"2026-07-05T06:18:45.157892+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.05052","created_at":"2026-07-05T06:18:45.157892+00:00"},{"alias_kind":"pith_short_12","alias_value":"HHMBAQWPGAEB","created_at":"2026-07-05T06:18:45.157892+00:00"},{"alias_kind":"pith_short_16","alias_value":"HHMBAQWPGAEBMBUM","created_at":"2026-07-05T06:18:45.157892+00:00"},{"alias_kind":"pith_short_8","alias_value":"HHMBAQWP","created_at":"2026-07-05T06:18:45.157892+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2401.02458","citing_title":"Data-Centric Foundation Models in Computational Healthcare: A Survey","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08045","citing_title":"Uncertainty-Aware Structured Data Extraction from Full CMR Reports via Distilled LLMs","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HHMBAQWPGAEBMBUMVPELHWB76S","json":"https://pith.science/pith/HHMBAQWPGAEBMBUMVPELHWB76S.json","graph_json":"https://pith.science/api/pith-number/HHMBAQWPGAEBMBUMVPELHWB76S/graph.json","events_json":"https://pith.science/api/pith-number/HHMBAQWPGAEBMBUMVPELHWB76S/events.json","paper":"https://pith.science/paper/HHMBAQWP"},"agent_actions":{"view_html":"https://pith.science/pith/HHMBAQWPGAEBMBUMVPELHWB76S","download_json":"https://pith.science/pith/HHMBAQWPGAEBMBUMVPELHWB76S.json","view_paper":"https://pith.science/paper/HHMBAQWP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.05052&json=true","fetch_graph":"https://pith.science/api/pith-number/HHMBAQWPGAEBMBUMVPELHWB76S/graph.json","fetch_events":"https://pith.science/api/pith-number/HHMBAQWPGAEBMBUMVPELHWB76S/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HHMBAQWPGAEBMBUMVPELHWB76S/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HHMBAQWPGAEBMBUMVPELHWB76S/action/storage_attestation","attest_author":"https://pith.science/pith/HHMBAQWPGAEBMBUMVPELHWB76S/action/author_attestation","sign_citation":"https://pith.science/pith/HHMBAQWPGAEBMBUMVPELHWB76S/action/citation_signature","submit_replication":"https://pith.science/pith/HHMBAQWPGAEBMBUMVPELHWB76S/action/replication_record"}},"created_at":"2026-07-05T06:18:45.157892+00:00","updated_at":"2026-07-05T06:18:45.157892+00:00"}