{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:HI4LEOLNRZNFXP23MQDIPE52LH","short_pith_number":"pith:HI4LEOLN","schema_version":"1.0","canonical_sha256":"3a38b2396d8e5a5bbf5b64068793ba59c7a81c347335127403b14bd055667a46","source":{"kind":"arxiv","id":"2305.01526","version":1},"attestation_state":"computed","paper":{"title":"Huatuo-26M, a Large-scale Chinese Medical QA Dataset","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Benyou Wang, Jianquan Li, Jie Fu, Prayag Tiwari, Xiangbo Wu, Xiang Wan, Xiaolong Xu, Xidong Wang, Zhiyi Zhang","submitted_at":"2023-05-02T15:33:01Z","abstract_excerpt":"In this paper, we release a largest ever medical Question Answering (QA) dataset with 26 million QA pairs. We benchmark many existing approaches in our dataset in terms of both retrieval and generation. Experimental results show that the existing models perform far lower than expected and the released dataset is still challenging in the pre-trained language model era. Moreover, we also experimentally show the benefit of the proposed dataset in many aspects: (i) trained models for other QA datasets in a zero-shot fashion; and (ii) as external knowledge for retrieval-augmented generation (RAG); "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.01526","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-05-02T15:33:01Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"41ff7b628d084917d4781d8744cd0122c6cbb6e14e0d2129830f030a29c3ecb8","abstract_canon_sha256":"60a55038082efa917fb57fa378acdfa6b0c30b616826fc9464309a058ddf17ac"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:06:26.779355Z","signature_b64":"t/p+GcSeek4Oewk3zmYFrOBcMfVwhvMntEsTdYjOx+EjHSk1u2et0KHWPBvcPbYQqyzN6ZYKmcUTUenEyuJkBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3a38b2396d8e5a5bbf5b64068793ba59c7a81c347335127403b14bd055667a46","last_reissued_at":"2026-07-05T06:06:26.778926Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:06:26.778926Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Huatuo-26M, a Large-scale Chinese Medical QA Dataset","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Benyou Wang, Jianquan Li, Jie Fu, Prayag Tiwari, Xiangbo Wu, Xiang Wan, Xiaolong Xu, Xidong Wang, Zhiyi Zhang","submitted_at":"2023-05-02T15:33:01Z","abstract_excerpt":"In this paper, we release a largest ever medical Question Answering (QA) dataset with 26 million QA pairs. We benchmark many existing approaches in our dataset in terms of both retrieval and generation. Experimental results show that the existing models perform far lower than expected and the released dataset is still challenging in the pre-trained language model era. Moreover, we also experimentally show the benefit of the proposed dataset in many aspects: (i) trained models for other QA datasets in a zero-shot fashion; and (ii) as external knowledge for retrieval-augmented generation (RAG); "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.01526","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.01526/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.01526","created_at":"2026-07-05T06:06:26.778985+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.01526v1","created_at":"2026-07-05T06:06:26.778985+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.01526","created_at":"2026-07-05T06:06:26.778985+00:00"},{"alias_kind":"pith_short_12","alias_value":"HI4LEOLNRZNF","created_at":"2026-07-05T06:06:26.778985+00:00"},{"alias_kind":"pith_short_16","alias_value":"HI4LEOLNRZNFXP23","created_at":"2026-07-05T06:06:26.778985+00:00"},{"alias_kind":"pith_short_8","alias_value":"HI4LEOLN","created_at":"2026-07-05T06:06:26.778985+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2401.02458","citing_title":"Data-Centric Foundation Models in Computational Healthcare: A Survey","ref_index":162,"is_internal_anchor":false},{"citing_arxiv_id":"2601.20375","citing_title":"LLM-AutoDP: Automatic Data Processing via LLM Agents for Model Fine-tuning","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HI4LEOLNRZNFXP23MQDIPE52LH","json":"https://pith.science/pith/HI4LEOLNRZNFXP23MQDIPE52LH.json","graph_json":"https://pith.science/api/pith-number/HI4LEOLNRZNFXP23MQDIPE52LH/graph.json","events_json":"https://pith.science/api/pith-number/HI4LEOLNRZNFXP23MQDIPE52LH/events.json","paper":"https://pith.science/paper/HI4LEOLN"},"agent_actions":{"view_html":"https://pith.science/pith/HI4LEOLNRZNFXP23MQDIPE52LH","download_json":"https://pith.science/pith/HI4LEOLNRZNFXP23MQDIPE52LH.json","view_paper":"https://pith.science/paper/HI4LEOLN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.01526&json=true","fetch_graph":"https://pith.science/api/pith-number/HI4LEOLNRZNFXP23MQDIPE52LH/graph.json","fetch_events":"https://pith.science/api/pith-number/HI4LEOLNRZNFXP23MQDIPE52LH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HI4LEOLNRZNFXP23MQDIPE52LH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HI4LEOLNRZNFXP23MQDIPE52LH/action/storage_attestation","attest_author":"https://pith.science/pith/HI4LEOLNRZNFXP23MQDIPE52LH/action/author_attestation","sign_citation":"https://pith.science/pith/HI4LEOLNRZNFXP23MQDIPE52LH/action/citation_signature","submit_replication":"https://pith.science/pith/HI4LEOLNRZNFXP23MQDIPE52LH/action/replication_record"}},"created_at":"2026-07-05T06:06:26.778985+00:00","updated_at":"2026-07-05T06:06:26.778985+00:00"}