{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:BF4Y2D274BU76WE7ZGMWPQC55G","short_pith_number":"pith:BF4Y2D27","schema_version":"1.0","canonical_sha256":"09798d0f5fe069ff589fc99967c05de98bb1574350ae21a07f683eb80df6912a","source":{"kind":"arxiv","id":"2501.07171","version":3},"attestation_state":"computed","paper":{"title":"BIOMEDICA: An Open Biomedical Image-Caption Archive, Dataset, and Vision-Language Models Derived from Scientific Literature","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Alejandro Lozano, Alfred Seunghoon Song, Anita Rau, Austin Wolfgang Katzer, Collin Chiu, Ivan Lopez, James Burgess, Jeffrey Gu, Jeffrey J Nirschl, Josiah Aklilu, Liangyu Chen, Min Woo Sun, Robert Tibshirani, Serena Yeung-Levy, Xiaohan Wang, Yuhui Zhang","submitted_at":"2025-01-13T09:58:03Z","abstract_excerpt":"The development of vision-language models (VLMs) is driven by large-scale and diverse multimodal datasets. However, progress toward generalist biomedical VLMs is limited by the lack of annotated, publicly accessible datasets across biology and medicine. Existing efforts are restricted to narrow domains, missing the full diversity of biomedical knowledge encoded in scientific literature. To address this gap, we introduce BIOMEDICA, a scalable, open-source framework to extract, annotate, and serialize the entirety of the PubMed Central Open Access subset into an easy-to-use, publicly accessible "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.07171","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-01-13T09:58:03Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"d55ff13c0fabf531127e2998b4668d159c697356b7e0e80c61e412dc2a863559","abstract_canon_sha256":"4853be36f1f70494cb67984c292c3a7faf948aa391604fce6ab3b15aba59cefb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:43:04.418133Z","signature_b64":"kLmxccoYizIit0dmQKVS5SZbqJrm9kZi1mRHvhkApEmE2aK52uLtOkaBGeR8H6vOH5cPaC4xqQgmwgg2bQR/AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"09798d0f5fe069ff589fc99967c05de98bb1574350ae21a07f683eb80df6912a","last_reissued_at":"2026-07-05T10:43:04.417660Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:43:04.417660Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BIOMEDICA: An Open Biomedical Image-Caption Archive, Dataset, and Vision-Language Models Derived from Scientific Literature","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Alejandro Lozano, Alfred Seunghoon Song, Anita Rau, Austin Wolfgang Katzer, Collin Chiu, Ivan Lopez, James Burgess, Jeffrey Gu, Jeffrey J Nirschl, Josiah Aklilu, Liangyu Chen, Min Woo Sun, Robert Tibshirani, Serena Yeung-Levy, Xiaohan Wang, Yuhui Zhang","submitted_at":"2025-01-13T09:58:03Z","abstract_excerpt":"The development of vision-language models (VLMs) is driven by large-scale and diverse multimodal datasets. However, progress toward generalist biomedical VLMs is limited by the lack of annotated, publicly accessible datasets across biology and medicine. Existing efforts are restricted to narrow domains, missing the full diversity of biomedical knowledge encoded in scientific literature. To address this gap, we introduce BIOMEDICA, a scalable, open-source framework to extract, annotate, and serialize the entirety of the PubMed Central Open Access subset into an easy-to-use, publicly accessible "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.07171","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.07171/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.07171","created_at":"2026-07-05T10:43:04.417710+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.07171v3","created_at":"2026-07-05T10:43:04.417710+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.07171","created_at":"2026-07-05T10:43:04.417710+00:00"},{"alias_kind":"pith_short_12","alias_value":"BF4Y2D274BU7","created_at":"2026-07-05T10:43:04.417710+00:00"},{"alias_kind":"pith_short_16","alias_value":"BF4Y2D274BU76WE7","created_at":"2026-07-05T10:43:04.417710+00:00"},{"alias_kind":"pith_short_8","alias_value":"BF4Y2D27","created_at":"2026-07-05T10:43:04.417710+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05436","citing_title":"Ten Headache Specialists versus Artificial Intelligence for Clinical Literature Summarization: A Critical Evaluation and Comparison","ref_index":124,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BF4Y2D274BU76WE7ZGMWPQC55G","json":"https://pith.science/pith/BF4Y2D274BU76WE7ZGMWPQC55G.json","graph_json":"https://pith.science/api/pith-number/BF4Y2D274BU76WE7ZGMWPQC55G/graph.json","events_json":"https://pith.science/api/pith-number/BF4Y2D274BU76WE7ZGMWPQC55G/events.json","paper":"https://pith.science/paper/BF4Y2D27"},"agent_actions":{"view_html":"https://pith.science/pith/BF4Y2D274BU76WE7ZGMWPQC55G","download_json":"https://pith.science/pith/BF4Y2D274BU76WE7ZGMWPQC55G.json","view_paper":"https://pith.science/paper/BF4Y2D27","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.07171&json=true","fetch_graph":"https://pith.science/api/pith-number/BF4Y2D274BU76WE7ZGMWPQC55G/graph.json","fetch_events":"https://pith.science/api/pith-number/BF4Y2D274BU76WE7ZGMWPQC55G/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BF4Y2D274BU76WE7ZGMWPQC55G/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BF4Y2D274BU76WE7ZGMWPQC55G/action/storage_attestation","attest_author":"https://pith.science/pith/BF4Y2D274BU76WE7ZGMWPQC55G/action/author_attestation","sign_citation":"https://pith.science/pith/BF4Y2D274BU76WE7ZGMWPQC55G/action/citation_signature","submit_replication":"https://pith.science/pith/BF4Y2D274BU76WE7ZGMWPQC55G/action/replication_record"}},"created_at":"2026-07-05T10:43:04.417710+00:00","updated_at":"2026-07-05T10:43:04.417710+00:00"}