{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:4HSQ7NMHDGID5URCJILQ3HBDK3","short_pith_number":"pith:4HSQ7NMH","schema_version":"1.0","canonical_sha256":"e1e50fb58719903ed2224a170d9c2356f07e5071d3a443bd867aa7158ddd253c","source":{"kind":"arxiv","id":"2312.15166","version":3},"attestation_state":"computed","paper":{"title":"SOLAR 10.7B: Scaling Large Language Models with Simple yet Effective Depth Up-Scaling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Changbae Ahn, Chanjun Park, Dahyun Kim, Gyoungjin Gim, Hwalsuk Lee, Hyeonju Lee, Hyeonwoo Kim, Hyunbyung Park, Jihoo Kim, Mikyoung Cha, Sanghoon Kim, Seonghoon Yang, Sukyung Lee, Sunghun Kim, Wonho Song, Wonsung Lee, Yungi Kim, Yunsu Kim","submitted_at":"2023-12-23T05:11:37Z","abstract_excerpt":"We introduce SOLAR 10.7B, a large language model (LLM) with 10.7 billion parameters, demonstrating superior performance in various natural language processing (NLP) tasks. Inspired by recent efforts to efficiently up-scale LLMs, we present a method for scaling LLMs called depth up-scaling (DUS), which encompasses depthwise scaling and continued pretraining. In contrast to other LLM up-scaling methods that use mixture-of-experts, DUS does not require complex changes to train and inference efficiently. We show experimentally that DUS is simple yet effective in scaling up high-performance LLMs fr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.15166","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-12-23T05:11:37Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"b5fd591e808d494d33029b19c9b6689e73bc217733b3a1cfe5e6269eda1d230d","abstract_canon_sha256":"a7ad845e043492c12242da96deb6f56a51643ebe8e4ade4400dd5d7481e6d9c1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:04:17.161992Z","signature_b64":"6ZMk4PvmKjxLo6D4Fs7h5B4kUwrR4nxzdUQZhxcURk01f5UBbAsE4wA48rxKiox6sl6crH8yHanx3ENveTI6Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e1e50fb58719903ed2224a170d9c2356f07e5071d3a443bd867aa7158ddd253c","last_reissued_at":"2026-07-05T08:04:17.161503Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:04:17.161503Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SOLAR 10.7B: Scaling Large Language Models with Simple yet Effective Depth Up-Scaling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Changbae Ahn, Chanjun Park, Dahyun Kim, Gyoungjin Gim, Hwalsuk Lee, Hyeonju Lee, Hyeonwoo Kim, Hyunbyung Park, Jihoo Kim, Mikyoung Cha, Sanghoon Kim, Seonghoon Yang, Sukyung Lee, Sunghun Kim, Wonho Song, Wonsung Lee, Yungi Kim, Yunsu Kim","submitted_at":"2023-12-23T05:11:37Z","abstract_excerpt":"We introduce SOLAR 10.7B, a large language model (LLM) with 10.7 billion parameters, demonstrating superior performance in various natural language processing (NLP) tasks. Inspired by recent efforts to efficiently up-scale LLMs, we present a method for scaling LLMs called depth up-scaling (DUS), which encompasses depthwise scaling and continued pretraining. In contrast to other LLM up-scaling methods that use mixture-of-experts, DUS does not require complex changes to train and inference efficiently. We show experimentally that DUS is simple yet effective in scaling up high-performance LLMs fr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.15166","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.15166/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.15166","created_at":"2026-07-05T08:04:17.161564+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.15166v3","created_at":"2026-07-05T08:04:17.161564+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.15166","created_at":"2026-07-05T08:04:17.161564+00:00"},{"alias_kind":"pith_short_12","alias_value":"4HSQ7NMHDGID","created_at":"2026-07-05T08:04:17.161564+00:00"},{"alias_kind":"pith_short_16","alias_value":"4HSQ7NMHDGID5URC","created_at":"2026-07-05T08:04:17.161564+00:00"},{"alias_kind":"pith_short_8","alias_value":"4HSQ7NMH","created_at":"2026-07-05T08:04:17.161564+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07404","citing_title":"Reversible Foundations: Training a 120B Sparse MoE through State-Preserving Scaling","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04909","citing_title":"BEATS: Bootstrapping E-commerce Attribute Taxonomies for Search through Iterative Human-AI Collaboration","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31796","citing_title":"CHERRY: Compressed Hierarchical Experts with Recurrent Representational Yield","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2404.10981","citing_title":"A Survey on Retrieval-Augmented Text Generation for Large Language Models","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2408.07666","citing_title":"Model Merging in LLMs, MLLMs, and Beyond: Methods, Theories, Applications and Opportunities","ref_index":109,"is_internal_anchor":false},{"citing_arxiv_id":"2512.13751","citing_title":"MIDUS: Memory-Infused Depth Up-Scaling","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2603.13224","citing_title":"Visual-ERM: Reward Modeling for Visual Equivalence","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2406.00515","citing_title":"A Survey on Large Language Models for Code Generation","ref_index":133,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05662","citing_title":"XL-SafetyBench: A Country-Grounded Cross-Cultural Benchmark for LLM Safety and Cultural Sensitivity","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12391","citing_title":"Chain-of-Models Pre-Training: Rethinking Training Acceleration of Vision Foundation Models","ref_index":30,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4HSQ7NMHDGID5URCJILQ3HBDK3","json":"https://pith.science/pith/4HSQ7NMHDGID5URCJILQ3HBDK3.json","graph_json":"https://pith.science/api/pith-number/4HSQ7NMHDGID5URCJILQ3HBDK3/graph.json","events_json":"https://pith.science/api/pith-number/4HSQ7NMHDGID5URCJILQ3HBDK3/events.json","paper":"https://pith.science/paper/4HSQ7NMH"},"agent_actions":{"view_html":"https://pith.science/pith/4HSQ7NMHDGID5URCJILQ3HBDK3","download_json":"https://pith.science/pith/4HSQ7NMHDGID5URCJILQ3HBDK3.json","view_paper":"https://pith.science/paper/4HSQ7NMH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.15166&json=true","fetch_graph":"https://pith.science/api/pith-number/4HSQ7NMHDGID5URCJILQ3HBDK3/graph.json","fetch_events":"https://pith.science/api/pith-number/4HSQ7NMHDGID5URCJILQ3HBDK3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4HSQ7NMHDGID5URCJILQ3HBDK3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4HSQ7NMHDGID5URCJILQ3HBDK3/action/storage_attestation","attest_author":"https://pith.science/pith/4HSQ7NMHDGID5URCJILQ3HBDK3/action/author_attestation","sign_citation":"https://pith.science/pith/4HSQ7NMHDGID5URCJILQ3HBDK3/action/citation_signature","submit_replication":"https://pith.science/pith/4HSQ7NMHDGID5URCJILQ3HBDK3/action/replication_record"}},"created_at":"2026-07-05T08:04:17.161564+00:00","updated_at":"2026-07-05T08:04:17.161564+00:00"}