{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:7FOQNN7R6BMQ3RHQV4LJCMVXPF","short_pith_number":"pith:7FOQNN7R","schema_version":"1.0","canonical_sha256":"f95d06b7f1f0590dc4f0af169132b7794cfea3afac68d5a533072c9e59034bc1","source":{"kind":"arxiv","id":"1912.12142","version":1},"attestation_state":"computed","paper":{"title":"Lung and Colon Cancer Histopathological Image Dataset (LC25000)","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","q-bio.QM"],"primary_cat":"eess.IV","authors_text":"Andrew A. Borkowski, Catherine P. Wilson, Lauren A. Deland, L. Brannon Thomas, Marilyn M. Bui, Stephen M. Mastorides","submitted_at":"2019-12-16T16:28:00Z","abstract_excerpt":"The field of Machine Learning, a subset of Artificial Intelligence, has led to remarkable advancements in many areas, including medicine. Machine Learning algorithms require large datasets to train computer models successfully. Although there are medical image datasets available, more image datasets are needed from a variety of medical entities, especially cancer pathology. Even more scarce are ML-ready image datasets. To address this need, we created an image dataset (LC25000) with 25,000 color images in 5 classes. Each class contains 5,000 images of the following histologic entities: colon a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1912.12142","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.IV","submitted_at":"2019-12-16T16:28:00Z","cross_cats_sorted":["cs.CV","q-bio.QM"],"title_canon_sha256":"307e27f4daf5bc071182628016ab09956e2a092a28a297494175122d933bdf0a","abstract_canon_sha256":"355ec19d2c38841bd2ed1f3c81942f79f958444697e95da67b0ce322d7d97407"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:28:40.650332Z","signature_b64":"Pb+cQ4Ki3w3F+MjFsW+Fqhobcj3cbwwokKwFWBPThTRgFBMjUYqYuh9/uUAa8L0fn05q0twnIXhBuflRkwc5BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f95d06b7f1f0590dc4f0af169132b7794cfea3afac68d5a533072c9e59034bc1","last_reissued_at":"2026-07-05T00:28:40.649790Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:28:40.649790Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Lung and Colon Cancer Histopathological Image Dataset (LC25000)","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","q-bio.QM"],"primary_cat":"eess.IV","authors_text":"Andrew A. Borkowski, Catherine P. Wilson, Lauren A. Deland, L. Brannon Thomas, Marilyn M. Bui, Stephen M. Mastorides","submitted_at":"2019-12-16T16:28:00Z","abstract_excerpt":"The field of Machine Learning, a subset of Artificial Intelligence, has led to remarkable advancements in many areas, including medicine. Machine Learning algorithms require large datasets to train computer models successfully. Although there are medical image datasets available, more image datasets are needed from a variety of medical entities, especially cancer pathology. Even more scarce are ML-ready image datasets. To address this need, we created an image dataset (LC25000) with 25,000 color images in 5 classes. Each class contains 5,000 images of the following histologic entities: colon a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1912.12142","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1912.12142/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1912.12142","created_at":"2026-07-05T00:28:40.649854+00:00"},{"alias_kind":"arxiv_version","alias_value":"1912.12142v1","created_at":"2026-07-05T00:28:40.649854+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1912.12142","created_at":"2026-07-05T00:28:40.649854+00:00"},{"alias_kind":"pith_short_12","alias_value":"7FOQNN7R6BMQ","created_at":"2026-07-05T00:28:40.649854+00:00"},{"alias_kind":"pith_short_16","alias_value":"7FOQNN7R6BMQ3RHQ","created_at":"2026-07-05T00:28:40.649854+00:00"},{"alias_kind":"pith_short_8","alias_value":"7FOQNN7R","created_at":"2026-07-05T00:28:40.649854+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07673","citing_title":"MedPMC: A Systematic Framework for Scaling High-Fidelity Medical Multimodal Data for Foundation Models","ref_index":95,"is_internal_anchor":true},{"citing_arxiv_id":"2606.24740","citing_title":"BioMedVR: Confusion-Aware Mixture-of-Prompt Experts for Biomedical Visual Reprogramming","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08126","citing_title":"One Stone, Three Birds: Self-adaptive Optimal Transport for Multi-VLM Selection, Adaptation, and Ensembling","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04922","citing_title":"Geometry-Aware Distillation for Prompt Tuning Biomedical Vision-Language Models","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25878","citing_title":"A Clinically Validated Foundation Model for Comprehensive Lung Pathology Interpretation","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2504.00470","citing_title":"Less is More: Efficient Black-box Attribution via Minimal Interpretable Subset Selection","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2601.01056","citing_title":"Enhancing Histopathological Image Classification via Integrated HOG and Deep Features with Robust Noise Performance","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03635","citing_title":"A Generative Foundation Model for Multimodal Histopathology","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2303.00915","citing_title":"BiomedCLIP: a multimodal biomedical foundation model pretrained from fifteen million scientific image-text pairs","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15711","citing_title":"SSMamba: A Self-Supervised Hybrid State Space Model for Pathological Image Classification","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16104","citing_title":"Dual-Modal Lung Cancer AI: Interpretable Radiology and Microscopy with Clinical Risk Integration","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7FOQNN7R6BMQ3RHQV4LJCMVXPF","json":"https://pith.science/pith/7FOQNN7R6BMQ3RHQV4LJCMVXPF.json","graph_json":"https://pith.science/api/pith-number/7FOQNN7R6BMQ3RHQV4LJCMVXPF/graph.json","events_json":"https://pith.science/api/pith-number/7FOQNN7R6BMQ3RHQV4LJCMVXPF/events.json","paper":"https://pith.science/paper/7FOQNN7R"},"agent_actions":{"view_html":"https://pith.science/pith/7FOQNN7R6BMQ3RHQV4LJCMVXPF","download_json":"https://pith.science/pith/7FOQNN7R6BMQ3RHQV4LJCMVXPF.json","view_paper":"https://pith.science/paper/7FOQNN7R","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1912.12142&json=true","fetch_graph":"https://pith.science/api/pith-number/7FOQNN7R6BMQ3RHQV4LJCMVXPF/graph.json","fetch_events":"https://pith.science/api/pith-number/7FOQNN7R6BMQ3RHQV4LJCMVXPF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7FOQNN7R6BMQ3RHQV4LJCMVXPF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7FOQNN7R6BMQ3RHQV4LJCMVXPF/action/storage_attestation","attest_author":"https://pith.science/pith/7FOQNN7R6BMQ3RHQV4LJCMVXPF/action/author_attestation","sign_citation":"https://pith.science/pith/7FOQNN7R6BMQ3RHQV4LJCMVXPF/action/citation_signature","submit_replication":"https://pith.science/pith/7FOQNN7R6BMQ3RHQV4LJCMVXPF/action/replication_record"}},"created_at":"2026-07-05T00:28:40.649854+00:00","updated_at":"2026-07-05T00:28:40.649854+00:00"}