{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:SKKEXQCK6B54NWNPU5OCTEJPAX","merge_version":"pith-open-graph-merge-v1","event_count":5,"valid_event_count":5,"invalid_event_count":0,"equivocation_count":1,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"226807d8055754f1641ff4cfa70746c91dd4a0de43b785d843152abfb180ec67","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-05-05T02:43:55Z","title_canon_sha256":"b5b3ae82ae781ac83b229a07ee1750e903abf3bc0e2d45ba25bc13cb1c9dcbd3"},"schema_version":"1.0","source":{"id":"2605.03301","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2605.03301","created_at":"2026-07-02T00:18:29Z"},{"alias_kind":"arxiv_version","alias_value":"2605.03301v2","created_at":"2026-07-02T00:18:29Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2605.03301","created_at":"2026-07-02T00:18:29Z"},{"alias_kind":"pith_short_12","alias_value":"SKKEXQCK6B54","created_at":"2026-07-02T00:18:29Z"},{"alias_kind":"pith_short_16","alias_value":"SKKEXQCK6B54NWNP","created_at":"2026-07-02T00:18:29Z"},{"alias_kind":"pith_short_8","alias_value":"SKKEXQCK","created_at":"2026-07-02T00:18:29Z"}],"graph_snapshots":[{"event_id":"sha256:9c1c40bb7e220c3992bf8cbad37c9b6ec9c49682a12ba4d47f10ec1831045162","target":"graph","created_at":"2026-07-02T00:18:29Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"Our best distilled model matches its teacher on structured PHI categories (DATE, DOCTOR, ID, PATIENT, PHONE) and achieves micro-averaged span-level precision of 0.88 and recall of 0.86 on standard workstation hardware."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"The set-cover diversity sampling combined with human-in-the-loop adjudication yields a dataset representative of modern clinical narratives that supports generalization beyond the sampled notes and institutions."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"SHIELD dataset and distilled DeBERTa v3 model achieve 0.88 micro precision and 0.86 recall on PHI de-identification while matching teacher performance on structured categories."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"Small language models distilled from large ones match teacher performance on structured patient identifiers in clinical notes at 0.88 precision and 0.86 recall on standard hardware."}],"snapshot_sha256":"2f4981214eb8dea3807e6b233db59756d31bcf555bedcf3f23ac4ae8940dfd78"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":false,"detectors_run":[{"findings_count":0,"name":"ai_meta_artifact","ran_at":"2026-05-20T14:34:18.584515Z","status":"completed","version":"1.0.0"},{"findings_count":0,"name":"doi_title_agreement","ran_at":"2026-05-20T01:31:21.515206Z","status":"completed","version":"1.0.0"},{"findings_count":3,"name":"doi_compliance","ran_at":"2026-05-19T15:29:22.348085Z","status":"completed","version":"1.0.0"}],"endpoint":"/pith/2605.03301/integrity.json","findings":[{"audited_at":"2026-05-19T15:29:22.348085Z","detected_arxiv_id":null,"detected_doi":"10.1101/2025.05.05.25326979v1","detector":"doi_compliance","finding_type":"unresolvable_identifier","note":"Identifier '10.1101/2025.05.05.25326979v1' is syntactically valid but the DOI registry (doi.org) returned 404, and Crossref / OpenAlex / internal corpus also have no record. The cited work could not be located through any authoritative source.","ref_index":22,"severity":"critical","verdict_class":"cross_source"},{"audited_at":"2026-05-19T15:29:22.348085Z","detected_arxiv_id":null,"detected_doi":"10.1101/2025.03.21.25323520v2","detector":"doi_compliance","finding_type":"unresolvable_identifier","note":"Identifier '10.1101/2025.03.21.25323520v2' is syntactically valid but the DOI registry (doi.org) returned 404, and Crossref / OpenAlex / internal corpus also have no record. The cited work could not be located through any authoritative source.","ref_index":10,"severity":"critical","verdict_class":"cross_source"},{"audited_at":"2026-05-19T15:29:22.348085Z","detected_arxiv_id":null,"detected_doi":"10.1038/s41591-024-03259-5","detector":"doi_compliance","finding_type":"unresolvable_identifier","note":"Identifier '10.1038/s41591-024-03259-5' is syntactically valid but the DOI registry (doi.org) returned 404, and Crossref / OpenAlex / internal corpus also have no record. The cited work could not be located through any authoritative source.","ref_index":17,"severity":"critical","verdict_class":"cross_source"}],"snapshot_sha256":"92ee8a7415ece0fc1bc77572ea201fb58b79759253acaabc677ffc6326e9621d","summary":{"advisory":0,"by_detector":{"doi_compliance":{"advisory":0,"critical":3,"informational":0,"total":3}},"critical":3,"informational":0}},"paper":{"abstract_excerpt":"De-identification of clinical text is a prerequisite for the secondary use of electronic health records. Existing public benchmarks such as the i2b2 2006 and 2014 corpora are over a decade old and lack the semantic and demographic diversity of modern clinical narratives. Large Language Models (LLMs) reach state-of-the-art zero-shot extraction, but their use at enterprise scale is limited by computational cost and by hospital data governance that restricts sending Protected Health Information (PHI) to cloud APIs. We introduce SHIELD (Synthetic Human-annotated Identifier-replaced Entries for Lea","authors_text":"David Love, Jose D. Posada, Priya Desai, Somalee Datta","cross_cats":["cs.AI"],"headline":"Small language models distilled from large ones match teacher performance on structured patient identifiers in clinical notes at 0.88 precision and 0.86 recall on standard hardware.","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-05-05T02:43:55Z","title":"SHIELD: A Diverse Clinical Note Dataset and Distilled Small Language Models for Enterprise-Scale De-identification"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2605.03301","kind":"arxiv","version":2},"verdict":{"created_at":"2026-05-07T16:48:41.935714Z","id":"4bf96010-e351-48ff-8184-c097b5653911","model_set":{"reader":"grok-4.3"},"one_line_summary":"SHIELD dataset and distilled DeBERTa v3 model achieve 0.88 micro precision and 0.86 recall on PHI de-identification while matching teacher performance on structured categories.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"Small language models distilled from large ones match teacher performance on structured patient identifiers in clinical notes at 0.88 precision and 0.86 recall on standard hardware.","strongest_claim":"Our best distilled model matches its teacher on structured PHI categories (DATE, DOCTOR, ID, PATIENT, PHONE) and achieves micro-averaged span-level precision of 0.88 and recall of 0.86 on standard workstation hardware.","weakest_assumption":"The set-cover diversity sampling combined with human-in-the-loop adjudication yields a dataset representative of modern clinical narratives that supports generalization beyond the sampled notes and institutions."}},"verdict_id":"4bf96010-e351-48ff-8184-c097b5653911"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:7450a77ba7d2ea22f595514a5b97e83ca22221ac9f767702520183d4a2804101","target":"record","created_at":"2026-07-02T00:18:29Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"226807d8055754f1641ff4cfa70746c91dd4a0de43b785d843152abfb180ec67","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-05-05T02:43:55Z","title_canon_sha256":"b5b3ae82ae781ac83b229a07ee1750e903abf3bc0e2d45ba25bc13cb1c9dcbd3"},"schema_version":"1.0","source":{"id":"2605.03301","kind":"arxiv","version":2}},"canonical_sha256":"92944bc04af07bc6d9afa75c29912f05ceec9288e8456ca80a8bb47e5bfd87c5","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"92944bc04af07bc6d9afa75c29912f05ceec9288e8456ca80a8bb47e5bfd87c5","first_computed_at":"2026-07-02T00:18:29.478279Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-02T00:18:29.478279Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"eJ3U1MPGUCFV2dNKB+HLos1KPcRkMfGASgQgo6wkefqkr+ctoJlg5ms4QhhpEh4RP63rtmi9JRvZjg6HggvbAw==","signature_status":"signed_v1","signed_at":"2026-07-02T00:18:29.478750Z","signed_message":"canonical_sha256_bytes"},"source_id":"2605.03301","source_kind":"arxiv","source_version":2}}},"equivocations":[{"signer_id":"pith.science","event_type":"integrity_finding","target":"integrity","event_ids":["sha256:00d448fa622961e3df52920453da1d60e438c61a7c8d33f3e3137000254011f7","sha256:2562ec9d081ca2828b284ff2614e2c5bed11072d80b459a4b0c84e74b97594fa","sha256:ca58763c3cd2c375685d18fe9530bc58052ab236813c1b89db44c7340028169a"]}],"invalid_events":[],"applied_event_ids":["sha256:7450a77ba7d2ea22f595514a5b97e83ca22221ac9f767702520183d4a2804101","sha256:9c1c40bb7e220c3992bf8cbad37c9b6ec9c49682a12ba4d47f10ec1831045162"],"state_sha256":"3f290f3c10d8c22f5a05b0e70261481a988436c05f693283166b759b60f3b9ab"}