{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:DFIXWE3SU4TCJ6DCOZKGC6TFI5","short_pith_number":"pith:DFIXWE3S","schema_version":"1.0","canonical_sha256":"19517b1372a72624f8627654617a654764212a5d87b965f1d2c628b15825af03","source":{"kind":"arxiv","id":"2311.09774","version":2},"attestation_state":"computed","paper":{"title":"HuatuoGPT-II, One-stage Training for Medical Adaption of LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Anningzhe Gao, Benyou Wang, Chuyi Kong, Dingjie Song, Feng Jiang, Haizhou Li, Hongbo Zhang, Jianquan Li, Junying Chen, Ke Ji, Shunian Chen, Wenya Xie, Xiang Wan, Xidong Wang","submitted_at":"2023-11-16T10:56:24Z","abstract_excerpt":"Adapting a language model into a specific domain, a.k.a `domain adaption', is a common practice when specialized knowledge, e.g. medicine, is not encapsulated in a general language model like Llama2. The challenge lies in the heterogeneity of data across the two training stages, as it varies in languages, genres, or formats. To tackle this and simplify the learning protocol, we propose to transform heterogeneous data, from the both pre-training and supervised stages, into a unified, simple input-output pair format. We validate the new protocol in the domains where proprietary LLMs like ChatGPT"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.09774","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-11-16T10:56:24Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"8102bfcfd837c5abed163c9b8ca104ce2e079b8fbe01c2af1a3d74419e643e1f","abstract_canon_sha256":"852f6919ee95c5a7bb30c6f7f2decdfd2888cc0cd72590a7911987d863ccf8dc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:06:56.058218Z","signature_b64":"o4ch1Y5HOgQ71r8g3+rXcgmfnYd/3/vf6bzOfVIPhkZCJu1mCXiyyGxvbAT8fpBfTBl5oES1as2Qxw+CsfQOCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"19517b1372a72624f8627654617a654764212a5d87b965f1d2c628b15825af03","last_reissued_at":"2026-07-05T09:06:56.057811Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:06:56.057811Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HuatuoGPT-II, One-stage Training for Medical Adaption of LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Anningzhe Gao, Benyou Wang, Chuyi Kong, Dingjie Song, Feng Jiang, Haizhou Li, Hongbo Zhang, Jianquan Li, Junying Chen, Ke Ji, Shunian Chen, Wenya Xie, Xiang Wan, Xidong Wang","submitted_at":"2023-11-16T10:56:24Z","abstract_excerpt":"Adapting a language model into a specific domain, a.k.a `domain adaption', is a common practice when specialized knowledge, e.g. medicine, is not encapsulated in a general language model like Llama2. The challenge lies in the heterogeneity of data across the two training stages, as it varies in languages, genres, or formats. To tackle this and simplify the learning protocol, we propose to transform heterogeneous data, from the both pre-training and supervised stages, into a unified, simple input-output pair format. We validate the new protocol in the domains where proprietary LLMs like ChatGPT"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.09774","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.09774/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.09774","created_at":"2026-07-05T09:06:56.057864+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.09774v2","created_at":"2026-07-05T09:06:56.057864+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.09774","created_at":"2026-07-05T09:06:56.057864+00:00"},{"alias_kind":"pith_short_12","alias_value":"DFIXWE3SU4TC","created_at":"2026-07-05T09:06:56.057864+00:00"},{"alias_kind":"pith_short_16","alias_value":"DFIXWE3SU4TCJ6DC","created_at":"2026-07-05T09:06:56.057864+00:00"},{"alias_kind":"pith_short_8","alias_value":"DFIXWE3S","created_at":"2026-07-05T09:06:56.057864+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08938","citing_title":"PACT: Learning Diverse Diagnostic Strategies via Privileged Synthesis and Branch Consensus","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03157","citing_title":"ClinicalMC: A Benchmark for Multi-Course Clinical Decision-Making with Large Language Models","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16215","citing_title":"Fully Open Meditron: An Auditable Pipeline for Clinical LLMs","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27860","citing_title":"C-MIG: Multi-view Information Gain-based Retrieval-Augmented Generation for Clinical Diagnosis Reasoning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11416","citing_title":"Freeze Deep, Train Shallow: Interpretable Layer Allocation for Continued Pre-Training","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16215","citing_title":"Fully Open Meditron: An Auditable Pipeline for Clinical LLMs","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2601.20375","citing_title":"LLM-AutoDP: Automatic Data Processing via LLM Agents for Model Fine-tuning","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2412.18925","citing_title":"HuatuoGPT-o1, Towards Medical Complex Reasoning with LLMs","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11416","citing_title":"Freeze Deep, Train Shallow: Interpretable Layer Allocation for Continued Pre-Training","ref_index":48,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DFIXWE3SU4TCJ6DCOZKGC6TFI5","json":"https://pith.science/pith/DFIXWE3SU4TCJ6DCOZKGC6TFI5.json","graph_json":"https://pith.science/api/pith-number/DFIXWE3SU4TCJ6DCOZKGC6TFI5/graph.json","events_json":"https://pith.science/api/pith-number/DFIXWE3SU4TCJ6DCOZKGC6TFI5/events.json","paper":"https://pith.science/paper/DFIXWE3S"},"agent_actions":{"view_html":"https://pith.science/pith/DFIXWE3SU4TCJ6DCOZKGC6TFI5","download_json":"https://pith.science/pith/DFIXWE3SU4TCJ6DCOZKGC6TFI5.json","view_paper":"https://pith.science/paper/DFIXWE3S","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.09774&json=true","fetch_graph":"https://pith.science/api/pith-number/DFIXWE3SU4TCJ6DCOZKGC6TFI5/graph.json","fetch_events":"https://pith.science/api/pith-number/DFIXWE3SU4TCJ6DCOZKGC6TFI5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DFIXWE3SU4TCJ6DCOZKGC6TFI5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DFIXWE3SU4TCJ6DCOZKGC6TFI5/action/storage_attestation","attest_author":"https://pith.science/pith/DFIXWE3SU4TCJ6DCOZKGC6TFI5/action/author_attestation","sign_citation":"https://pith.science/pith/DFIXWE3SU4TCJ6DCOZKGC6TFI5/action/citation_signature","submit_replication":"https://pith.science/pith/DFIXWE3SU4TCJ6DCOZKGC6TFI5/action/replication_record"}},"created_at":"2026-07-05T09:06:56.057864+00:00","updated_at":"2026-07-05T09:06:56.057864+00:00"}