{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:LYV7NO4E55SMOVVI5KOEWOXOSL","short_pith_number":"pith:LYV7NO4E","schema_version":"1.0","canonical_sha256":"5e2bf6bb84ef64c756a8ea9c4b3aee92c716213fe09762f47ecd580b75624958","source":{"kind":"arxiv","id":"2004.09167","version":3},"attestation_state":"computed","paper":{"title":"CheXbert: Combining Automatic Labelers and Expert Annotations for Accurate Radiology Report Labeling Using BERT","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.IR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Akshay Smit, Andrew Y. Ng, Anuj Pareek, Matthew P. Lungren, Pranav Rajpurkar, Saahil Jain","submitted_at":"2020-04-20T09:46:40Z","abstract_excerpt":"The extraction of labels from radiology text reports enables large-scale training of medical imaging models. Existing approaches to report labeling typically rely either on sophisticated feature engineering based on medical domain knowledge or manual annotations by experts. In this work, we introduce a BERT-based approach to medical image report labeling that exploits both the scale of available rule-based systems and the quality of expert annotations. We demonstrate superior performance of a biomedically pretrained BERT model first trained on annotations of a rule-based labeler and then finet"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2004.09167","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-04-20T09:46:40Z","cross_cats_sorted":["cs.IR","cs.LG"],"title_canon_sha256":"38282268c83ad89a744a3919d5726baf4d30d807f4070955829d4cfcdfe8046c","abstract_canon_sha256":"48e35ed9b2cf2147943e5d22d26763cdbaf8692484aa5a1a27a6bb79042ea8ef"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:43:56.450834Z","signature_b64":"zrves6W11n6E6tge7fruTaZB3CPyKzPOhFHYbG3CRE8ItzdSAC2WP4lduJt9+ysJkG7BM5XTIIhisGcIG4B/DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5e2bf6bb84ef64c756a8ea9c4b3aee92c716213fe09762f47ecd580b75624958","last_reissued_at":"2026-07-05T01:43:56.450270Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:43:56.450270Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CheXbert: Combining Automatic Labelers and Expert Annotations for Accurate Radiology Report Labeling Using BERT","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.IR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Akshay Smit, Andrew Y. Ng, Anuj Pareek, Matthew P. Lungren, Pranav Rajpurkar, Saahil Jain","submitted_at":"2020-04-20T09:46:40Z","abstract_excerpt":"The extraction of labels from radiology text reports enables large-scale training of medical imaging models. Existing approaches to report labeling typically rely either on sophisticated feature engineering based on medical domain knowledge or manual annotations by experts. In this work, we introduce a BERT-based approach to medical image report labeling that exploits both the scale of available rule-based systems and the quality of expert annotations. We demonstrate superior performance of a biomedically pretrained BERT model first trained on annotations of a rule-based labeler and then finet"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2004.09167","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2004.09167/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2004.09167","created_at":"2026-07-05T01:43:56.450360+00:00"},{"alias_kind":"arxiv_version","alias_value":"2004.09167v3","created_at":"2026-07-05T01:43:56.450360+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2004.09167","created_at":"2026-07-05T01:43:56.450360+00:00"},{"alias_kind":"pith_short_12","alias_value":"LYV7NO4E55SM","created_at":"2026-07-05T01:43:56.450360+00:00"},{"alias_kind":"pith_short_16","alias_value":"LYV7NO4E55SMOVVI","created_at":"2026-07-05T01:43:56.450360+00:00"},{"alias_kind":"pith_short_8","alias_value":"LYV7NO4E","created_at":"2026-07-05T01:43:56.450360+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21447","citing_title":"Precision Recall Controllable Radiology Report Generation via Hybrid Natural Language and Clinical Reward Learning","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21447","citing_title":"Precision Recall Controllable Radiology Report Generation via Hybrid Natural Language and Clinical Reward Learning","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10061","citing_title":"BenSyc: Benchmarking Conversational Sycophancy and Human Alignment in LLMs for Bengali Contexts","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2408.16213","citing_title":"M4CXR: Exploring Multi-task Potentials of Multi-modal Large Language Models for Chest X-ray Interpretation","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2509.04471","citing_title":"MOSAIC: A Multilingual, Taxonomy-Agnostic, and Computationally Efficient Approach for Radiological Report Classification","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17071","citing_title":"AnchorDiff: Topology-Aware Masked Diffusion with Confidence-based Rewriting for Radiology Report Generation","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2509.22258","citing_title":"Beyond Classification Accuracy: Neural-MedBench and the Need for Deeper Reasoning Benchmarks","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11304","citing_title":"CheXTemporal: A Dataset for Temporally-Grounded Reasoning in Chest Radiography","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24001","citing_title":"CT-FineBench: A Diagnostic Fidelity Benchmark for Fine-Grained Evaluation of CT Report Generation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22989","citing_title":"CheXmix: Unified Generative Pretraining for Vision Language Models in Medical Imaging","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00421","citing_title":"RadLite: Multi-Task LoRA Fine-Tuning of Small Language Models for CPU-Deployable Radiology AI","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08600","citing_title":"Gaze2Report: Radiology Report Generation via Visual-Gaze Prompt Tuning of LLMs","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13598","citing_title":"Enhancing Reinforcement Learning for Radiology Report Generation with Evidence-aware Rewards and Self-correcting Preference Learning","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LYV7NO4E55SMOVVI5KOEWOXOSL","json":"https://pith.science/pith/LYV7NO4E55SMOVVI5KOEWOXOSL.json","graph_json":"https://pith.science/api/pith-number/LYV7NO4E55SMOVVI5KOEWOXOSL/graph.json","events_json":"https://pith.science/api/pith-number/LYV7NO4E55SMOVVI5KOEWOXOSL/events.json","paper":"https://pith.science/paper/LYV7NO4E"},"agent_actions":{"view_html":"https://pith.science/pith/LYV7NO4E55SMOVVI5KOEWOXOSL","download_json":"https://pith.science/pith/LYV7NO4E55SMOVVI5KOEWOXOSL.json","view_paper":"https://pith.science/paper/LYV7NO4E","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2004.09167&json=true","fetch_graph":"https://pith.science/api/pith-number/LYV7NO4E55SMOVVI5KOEWOXOSL/graph.json","fetch_events":"https://pith.science/api/pith-number/LYV7NO4E55SMOVVI5KOEWOXOSL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LYV7NO4E55SMOVVI5KOEWOXOSL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LYV7NO4E55SMOVVI5KOEWOXOSL/action/storage_attestation","attest_author":"https://pith.science/pith/LYV7NO4E55SMOVVI5KOEWOXOSL/action/author_attestation","sign_citation":"https://pith.science/pith/LYV7NO4E55SMOVVI5KOEWOXOSL/action/citation_signature","submit_replication":"https://pith.science/pith/LYV7NO4E55SMOVVI5KOEWOXOSL/action/replication_record"}},"created_at":"2026-07-05T01:43:56.450360+00:00","updated_at":"2026-07-05T01:43:56.450360+00:00"}