{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CU2BUEJWYFRGG4GFSEPGZJEXGZ","short_pith_number":"pith:CU2BUEJW","schema_version":"1.0","canonical_sha256":"15341a1136c1626370c5911e6ca4973675bdc62a98bcdef7573c70d9014759a3","source":{"kind":"arxiv","id":"2507.15275","version":1},"attestation_state":"computed","paper":{"title":"ChiMed 2.0: Advancing Chinese Medical Dataset in Facilitating Large Language Modeling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Junjie Liu, Yan Song, Yuanhe Tian, Yuxiang Li, Zhizhou Kou","submitted_at":"2025-07-21T06:23:16Z","abstract_excerpt":"Building high-quality data resources is crucial for advancing artificial intelligence research and applications in specific domains, particularly in the Chinese medical domain. Existing Chinese medical datasets are limited in size and narrow in domain coverage, falling short of the diverse corpora required for effective pre-training. Moreover, most datasets are designed solely for LLM fine-tuning and do not support pre-training and reinforcement learning from human feedback (RLHF). In this paper, we propose a Chinese medical dataset named ChiMed 2.0, which extends our previous work ChiMed, and"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.15275","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-07-21T06:23:16Z","cross_cats_sorted":[],"title_canon_sha256":"9d1c72e54c909a5f6e8469549e7decc7af88efa0baa041a799807cbf15ec2e40","abstract_canon_sha256":"44a8b0ab660dbfb27d15a6f105daf30d0e5a0ac7c19c5e23dba6462382e6a391"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:40:21.715861Z","signature_b64":"fGlF2AVLPlshahVnMXMJrlXjjQ9FC3AYyIFQKPZ/jB9iy/Poyp/IWMN+OqwtsOTpoFkCcfoYF+GF7O+KwWBDDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"15341a1136c1626370c5911e6ca4973675bdc62a98bcdef7573c70d9014759a3","last_reissued_at":"2026-07-05T11:40:21.715342Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:40:21.715342Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ChiMed 2.0: Advancing Chinese Medical Dataset in Facilitating Large Language Modeling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Junjie Liu, Yan Song, Yuanhe Tian, Yuxiang Li, Zhizhou Kou","submitted_at":"2025-07-21T06:23:16Z","abstract_excerpt":"Building high-quality data resources is crucial for advancing artificial intelligence research and applications in specific domains, particularly in the Chinese medical domain. Existing Chinese medical datasets are limited in size and narrow in domain coverage, falling short of the diverse corpora required for effective pre-training. Moreover, most datasets are designed solely for LLM fine-tuning and do not support pre-training and reinforcement learning from human feedback (RLHF). In this paper, we propose a Chinese medical dataset named ChiMed 2.0, which extends our previous work ChiMed, and"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.15275","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.15275/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.15275","created_at":"2026-07-05T11:40:21.715404+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.15275v1","created_at":"2026-07-05T11:40:21.715404+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.15275","created_at":"2026-07-05T11:40:21.715404+00:00"},{"alias_kind":"pith_short_12","alias_value":"CU2BUEJWYFRG","created_at":"2026-07-05T11:40:21.715404+00:00"},{"alias_kind":"pith_short_16","alias_value":"CU2BUEJWYFRGG4GF","created_at":"2026-07-05T11:40:21.715404+00:00"},{"alias_kind":"pith_short_8","alias_value":"CU2BUEJW","created_at":"2026-07-05T11:40:21.715404+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CU2BUEJWYFRGG4GFSEPGZJEXGZ","json":"https://pith.science/pith/CU2BUEJWYFRGG4GFSEPGZJEXGZ.json","graph_json":"https://pith.science/api/pith-number/CU2BUEJWYFRGG4GFSEPGZJEXGZ/graph.json","events_json":"https://pith.science/api/pith-number/CU2BUEJWYFRGG4GFSEPGZJEXGZ/events.json","paper":"https://pith.science/paper/CU2BUEJW"},"agent_actions":{"view_html":"https://pith.science/pith/CU2BUEJWYFRGG4GFSEPGZJEXGZ","download_json":"https://pith.science/pith/CU2BUEJWYFRGG4GFSEPGZJEXGZ.json","view_paper":"https://pith.science/paper/CU2BUEJW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.15275&json=true","fetch_graph":"https://pith.science/api/pith-number/CU2BUEJWYFRGG4GFSEPGZJEXGZ/graph.json","fetch_events":"https://pith.science/api/pith-number/CU2BUEJWYFRGG4GFSEPGZJEXGZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CU2BUEJWYFRGG4GFSEPGZJEXGZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CU2BUEJWYFRGG4GFSEPGZJEXGZ/action/storage_attestation","attest_author":"https://pith.science/pith/CU2BUEJWYFRGG4GFSEPGZJEXGZ/action/author_attestation","sign_citation":"https://pith.science/pith/CU2BUEJWYFRGG4GFSEPGZJEXGZ/action/citation_signature","submit_replication":"https://pith.science/pith/CU2BUEJWYFRGG4GFSEPGZJEXGZ/action/replication_record"}},"created_at":"2026-07-05T11:40:21.715404+00:00","updated_at":"2026-07-05T11:40:21.715404+00:00"}