{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:B6DBS43RZDN4PDPIYY3DX3AKDN","short_pith_number":"pith:B6DBS43R","schema_version":"1.0","canonical_sha256":"0f86197371c8dbc78de8c6363bec0a1b52422267ff438e43dd3835b8e3dacc1e","source":{"kind":"arxiv","id":"2311.03301","version":2},"attestation_state":"computed","paper":{"title":"Ziya2: Data-centric Learning is All LLMs Need","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dixiang Zhang, Hao Wang, Jiaxing Zhang, Junqing He, Junyu Lu, Kunhao Pan, Ping Yang, Qi Yang, Renliang Sun, Ruyi Gan, Xiaojun Wu, Yan Song, Yuanhe Tian, Ziwei Wu","submitted_at":"2023-11-06T17:49:34Z","abstract_excerpt":"Various large language models (LLMs) have been proposed in recent years, including closed- and open-source ones, continually setting new records on multiple benchmarks. However, the development of LLMs still faces several issues, such as high cost of training models from scratch, and continual pre-training leading to catastrophic forgetting, etc. Although many such issues are addressed along the line of research on LLMs, an important yet practical limitation is that many studies overly pursue enlarging model sizes without comprehensively analyzing and optimizing the use of pre-training data in"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.03301","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-11-06T17:49:34Z","cross_cats_sorted":[],"title_canon_sha256":"2d0b49a0bcf4cd149fbb920ee1012fe08714b013e600abf96bf69772cc7164a9","abstract_canon_sha256":"dac15f7088fe4ee33fa1ef3f7515230d4b760ab2e9e5fc82e44703344addd8aa"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:04:15.776269Z","signature_b64":"e60Df3U+z3T2tEfmQG+3w+7XwBKzW7l36ji5A2/qUYDpABix1HmvMdlCzjfaz76VuSLRLcwt6DKwQ7fMyIekDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0f86197371c8dbc78de8c6363bec0a1b52422267ff438e43dd3835b8e3dacc1e","last_reissued_at":"2026-07-05T08:04:15.775838Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:04:15.775838Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Ziya2: Data-centric Learning is All LLMs Need","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dixiang Zhang, Hao Wang, Jiaxing Zhang, Junqing He, Junyu Lu, Kunhao Pan, Ping Yang, Qi Yang, Renliang Sun, Ruyi Gan, Xiaojun Wu, Yan Song, Yuanhe Tian, Ziwei Wu","submitted_at":"2023-11-06T17:49:34Z","abstract_excerpt":"Various large language models (LLMs) have been proposed in recent years, including closed- and open-source ones, continually setting new records on multiple benchmarks. However, the development of LLMs still faces several issues, such as high cost of training models from scratch, and continual pre-training leading to catastrophic forgetting, etc. Although many such issues are addressed along the line of research on LLMs, an important yet practical limitation is that many studies overly pursue enlarging model sizes without comprehensively analyzing and optimizing the use of pre-training data in"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.03301","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.03301/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.03301","created_at":"2026-07-05T08:04:15.775903+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.03301v2","created_at":"2026-07-05T08:04:15.775903+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.03301","created_at":"2026-07-05T08:04:15.775903+00:00"},{"alias_kind":"pith_short_12","alias_value":"B6DBS43RZDN4","created_at":"2026-07-05T08:04:15.775903+00:00"},{"alias_kind":"pith_short_16","alias_value":"B6DBS43RZDN4PDPI","created_at":"2026-07-05T08:04:15.775903+00:00"},{"alias_kind":"pith_short_8","alias_value":"B6DBS43R","created_at":"2026-07-05T08:04:15.775903+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/B6DBS43RZDN4PDPIYY3DX3AKDN","json":"https://pith.science/pith/B6DBS43RZDN4PDPIYY3DX3AKDN.json","graph_json":"https://pith.science/api/pith-number/B6DBS43RZDN4PDPIYY3DX3AKDN/graph.json","events_json":"https://pith.science/api/pith-number/B6DBS43RZDN4PDPIYY3DX3AKDN/events.json","paper":"https://pith.science/paper/B6DBS43R"},"agent_actions":{"view_html":"https://pith.science/pith/B6DBS43RZDN4PDPIYY3DX3AKDN","download_json":"https://pith.science/pith/B6DBS43RZDN4PDPIYY3DX3AKDN.json","view_paper":"https://pith.science/paper/B6DBS43R","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.03301&json=true","fetch_graph":"https://pith.science/api/pith-number/B6DBS43RZDN4PDPIYY3DX3AKDN/graph.json","fetch_events":"https://pith.science/api/pith-number/B6DBS43RZDN4PDPIYY3DX3AKDN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/B6DBS43RZDN4PDPIYY3DX3AKDN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/B6DBS43RZDN4PDPIYY3DX3AKDN/action/storage_attestation","attest_author":"https://pith.science/pith/B6DBS43RZDN4PDPIYY3DX3AKDN/action/author_attestation","sign_citation":"https://pith.science/pith/B6DBS43RZDN4PDPIYY3DX3AKDN/action/citation_signature","submit_replication":"https://pith.science/pith/B6DBS43RZDN4PDPIYY3DX3AKDN/action/replication_record"}},"created_at":"2026-07-05T08:04:15.775903+00:00","updated_at":"2026-07-05T08:04:15.775903+00:00"}