{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PBTCRJDZZQFEVN6IWMMCNBT33R","short_pith_number":"pith:PBTCRJDZ","schema_version":"1.0","canonical_sha256":"786628a479cc0a4ab7c8b31826867bdc55e8150d1773c49491afa6f970ceb2d4","source":{"kind":"arxiv","id":"2408.00588","version":1},"attestation_state":"computed","paper":{"title":"Closing the gap between open-source and commercial large language models for medical evidence summarization","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Ali Soroush, Betina R. Idnay, Chunhua Weng, Elizabeth Park, Gongbo Zhang, Jordan G. Nestor, Matthew E. Spotnitz, Qiao Jin, Song Wang, Thomas Campion, Yifan Peng, Yiliang Zhou, Yiming Luo, Zhiyong Lu","submitted_at":"2024-07-25T05:03:01Z","abstract_excerpt":"Large language models (LLMs) hold great promise in summarizing medical evidence. Most recent studies focus on the application of proprietary LLMs. Using proprietary LLMs introduces multiple risk factors, including a lack of transparency and vendor dependency. While open-source LLMs allow better transparency and customization, their performance falls short compared to proprietary ones. In this study, we investigated to what extent fine-tuning open-source LLMs can further improve their performance in summarizing medical evidence. Utilizing a benchmark dataset, MedReview, consisting of 8,161 pair"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.00588","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-25T05:03:01Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"1bab2ef88472ce1885baec97f390e205739b797d9057e7fdcba36d67260c9140","abstract_canon_sha256":"2da154c329451fcb3ecc32ee04dcbcbecf314daaf6222b262c22a06fc9115b13"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:51:06.254475Z","signature_b64":"ur+bEBoc3Fze1rayuinwYSK7XJfl5AiBQ0KLosRCU27jdIXgJXgWFBnjXgxmYatYQMZQPg+HVfQSOemz46QAAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"786628a479cc0a4ab7c8b31826867bdc55e8150d1773c49491afa6f970ceb2d4","last_reissued_at":"2026-07-05T08:51:06.254069Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:51:06.254069Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Closing the gap between open-source and commercial large language models for medical evidence summarization","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Ali Soroush, Betina R. Idnay, Chunhua Weng, Elizabeth Park, Gongbo Zhang, Jordan G. Nestor, Matthew E. Spotnitz, Qiao Jin, Song Wang, Thomas Campion, Yifan Peng, Yiliang Zhou, Yiming Luo, Zhiyong Lu","submitted_at":"2024-07-25T05:03:01Z","abstract_excerpt":"Large language models (LLMs) hold great promise in summarizing medical evidence. Most recent studies focus on the application of proprietary LLMs. Using proprietary LLMs introduces multiple risk factors, including a lack of transparency and vendor dependency. While open-source LLMs allow better transparency and customization, their performance falls short compared to proprietary ones. In this study, we investigated to what extent fine-tuning open-source LLMs can further improve their performance in summarizing medical evidence. Utilizing a benchmark dataset, MedReview, consisting of 8,161 pair"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.00588","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.00588/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.00588","created_at":"2026-07-05T08:51:06.254126+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.00588v1","created_at":"2026-07-05T08:51:06.254126+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.00588","created_at":"2026-07-05T08:51:06.254126+00:00"},{"alias_kind":"pith_short_12","alias_value":"PBTCRJDZZQFE","created_at":"2026-07-05T08:51:06.254126+00:00"},{"alias_kind":"pith_short_16","alias_value":"PBTCRJDZZQFEVN6I","created_at":"2026-07-05T08:51:06.254126+00:00"},{"alias_kind":"pith_short_8","alias_value":"PBTCRJDZ","created_at":"2026-07-05T08:51:06.254126+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.20598","citing_title":"Fine-Tuning and Prompt Engineering of LLMs, for the Creation of Multi-Agent AI for Addressing Sustainable Protein Production Challenges","ref_index":34,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PBTCRJDZZQFEVN6IWMMCNBT33R","json":"https://pith.science/pith/PBTCRJDZZQFEVN6IWMMCNBT33R.json","graph_json":"https://pith.science/api/pith-number/PBTCRJDZZQFEVN6IWMMCNBT33R/graph.json","events_json":"https://pith.science/api/pith-number/PBTCRJDZZQFEVN6IWMMCNBT33R/events.json","paper":"https://pith.science/paper/PBTCRJDZ"},"agent_actions":{"view_html":"https://pith.science/pith/PBTCRJDZZQFEVN6IWMMCNBT33R","download_json":"https://pith.science/pith/PBTCRJDZZQFEVN6IWMMCNBT33R.json","view_paper":"https://pith.science/paper/PBTCRJDZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.00588&json=true","fetch_graph":"https://pith.science/api/pith-number/PBTCRJDZZQFEVN6IWMMCNBT33R/graph.json","fetch_events":"https://pith.science/api/pith-number/PBTCRJDZZQFEVN6IWMMCNBT33R/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PBTCRJDZZQFEVN6IWMMCNBT33R/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PBTCRJDZZQFEVN6IWMMCNBT33R/action/storage_attestation","attest_author":"https://pith.science/pith/PBTCRJDZZQFEVN6IWMMCNBT33R/action/author_attestation","sign_citation":"https://pith.science/pith/PBTCRJDZZQFEVN6IWMMCNBT33R/action/citation_signature","submit_replication":"https://pith.science/pith/PBTCRJDZZQFEVN6IWMMCNBT33R/action/replication_record"}},"created_at":"2026-07-05T08:51:06.254126+00:00","updated_at":"2026-07-05T08:51:06.254126+00:00"}