{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:IOS6RK5VTXGDTUFO4S53IWSBRS","short_pith_number":"pith:IOS6RK5V","schema_version":"1.0","canonical_sha256":"43a5e8abb59dcc39d0aee4bbb45a418c93d6728f51addb42db0ed17e4721df48","source":{"kind":"arxiv","id":"2306.11207","version":4},"attestation_state":"computed","paper":{"title":"Quilt-1M: One Million Image-Text Pairs for Histopathology","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Dylan Stefan Chan Geva, Fatemeh Ghezloo, Fatwir Sheikh Mohammed, Linda Shapiro, Mehmet Saygin Seyfioglu, Pavan Kumar Anand, Ranjay Krishna, Wisdom Oluchi Ikezogwo","submitted_at":"2023-06-20T00:14:47Z","abstract_excerpt":"Recent accelerations in multi-modal applications have been made possible with the plethora of image and text data available online. However, the scarcity of analogous data in the medical field, specifically in histopathology, has slowed comparable progress. To enable similar representation learning for histopathology, we turn to YouTube, an untapped resource of videos, offering $1,087$ hours of valuable educational histopathology videos from expert clinicians. From YouTube, we curate QUILT: a large-scale vision-language dataset consisting of $802, 144$ image and text pairs. QUILT was automatic"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.11207","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-06-20T00:14:47Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"c39692aba40ffaf378622aab58578805fddac9e8066d31767a76c5a73c76d6f4","abstract_canon_sha256":"48131c0d9a020fa33b2043fd414c1814f421db72f5f7978e7c13aa925747accf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:00:25.526377Z","signature_b64":"m+gLpszjNKfXpCxwYxGmqwAbknc+A4nrfuyM9/WySSM6FwKB3+BRSsl4P+zEcrjFS8x2684Xlw+t2qEAiYi9Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"43a5e8abb59dcc39d0aee4bbb45a418c93d6728f51addb42db0ed17e4721df48","last_reissued_at":"2026-07-05T10:00:25.525873Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:00:25.525873Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Quilt-1M: One Million Image-Text Pairs for Histopathology","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Dylan Stefan Chan Geva, Fatemeh Ghezloo, Fatwir Sheikh Mohammed, Linda Shapiro, Mehmet Saygin Seyfioglu, Pavan Kumar Anand, Ranjay Krishna, Wisdom Oluchi Ikezogwo","submitted_at":"2023-06-20T00:14:47Z","abstract_excerpt":"Recent accelerations in multi-modal applications have been made possible with the plethora of image and text data available online. However, the scarcity of analogous data in the medical field, specifically in histopathology, has slowed comparable progress. To enable similar representation learning for histopathology, we turn to YouTube, an untapped resource of videos, offering $1,087$ hours of valuable educational histopathology videos from expert clinicians. From YouTube, we curate QUILT: a large-scale vision-language dataset consisting of $802, 144$ image and text pairs. QUILT was automatic"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.11207","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.11207/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.11207","created_at":"2026-07-05T10:00:25.525933+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.11207v4","created_at":"2026-07-05T10:00:25.525933+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.11207","created_at":"2026-07-05T10:00:25.525933+00:00"},{"alias_kind":"pith_short_12","alias_value":"IOS6RK5VTXGD","created_at":"2026-07-05T10:00:25.525933+00:00"},{"alias_kind":"pith_short_16","alias_value":"IOS6RK5VTXGDTUFO","created_at":"2026-07-05T10:00:25.525933+00:00"},{"alias_kind":"pith_short_8","alias_value":"IOS6RK5V","created_at":"2026-07-05T10:00:25.525933+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28697","citing_title":"Mitigating Batch Effects in Histopathology via Language-Mediated Robust Embedding Generation","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2401.02458","citing_title":"Data-Centric Foundation Models in Computational Healthcare: A Survey","ref_index":124,"is_internal_anchor":false},{"citing_arxiv_id":"2303.00915","citing_title":"BiomedCLIP: a multimodal biomedical foundation model pretrained from fifteen million scientific image-text pairs","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17570","citing_title":"PBSBench: A Multi-Level Vision-Language Framework and Benchmark for Hematopathology Whole Slide Image Interpretation","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IOS6RK5VTXGDTUFO4S53IWSBRS","json":"https://pith.science/pith/IOS6RK5VTXGDTUFO4S53IWSBRS.json","graph_json":"https://pith.science/api/pith-number/IOS6RK5VTXGDTUFO4S53IWSBRS/graph.json","events_json":"https://pith.science/api/pith-number/IOS6RK5VTXGDTUFO4S53IWSBRS/events.json","paper":"https://pith.science/paper/IOS6RK5V"},"agent_actions":{"view_html":"https://pith.science/pith/IOS6RK5VTXGDTUFO4S53IWSBRS","download_json":"https://pith.science/pith/IOS6RK5VTXGDTUFO4S53IWSBRS.json","view_paper":"https://pith.science/paper/IOS6RK5V","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.11207&json=true","fetch_graph":"https://pith.science/api/pith-number/IOS6RK5VTXGDTUFO4S53IWSBRS/graph.json","fetch_events":"https://pith.science/api/pith-number/IOS6RK5VTXGDTUFO4S53IWSBRS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IOS6RK5VTXGDTUFO4S53IWSBRS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IOS6RK5VTXGDTUFO4S53IWSBRS/action/storage_attestation","attest_author":"https://pith.science/pith/IOS6RK5VTXGDTUFO4S53IWSBRS/action/author_attestation","sign_citation":"https://pith.science/pith/IOS6RK5VTXGDTUFO4S53IWSBRS/action/citation_signature","submit_replication":"https://pith.science/pith/IOS6RK5VTXGDTUFO4S53IWSBRS/action/replication_record"}},"created_at":"2026-07-05T10:00:25.525933+00:00","updated_at":"2026-07-05T10:00:25.525933+00:00"}