{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GP4L2ZNGYYAUDWF7PR6RTRO3O3","short_pith_number":"pith:GP4L2ZNG","schema_version":"1.0","canonical_sha256":"33f8bd65a6c60141d8bf7c7d19c5db76e434ca2c3b957990119d549d00ce344d","source":{"kind":"arxiv","id":"2410.15226","version":2},"attestation_state":"computed","paper":{"title":"On the Diversity of Synthetic Data and its Impact on Training Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Abdul Waheed, Bhiksha Raj, Hao Chen, Jindong Wang, Marah I. Abdin, Xiang Li, Yidong Wang","submitted_at":"2024-10-19T22:14:07Z","abstract_excerpt":"The rise of Large Language Models (LLMs) has accentuated the need for diverse, high-quality pre-training data. Synthetic data emerges as a viable solution to the challenges of data scarcity and inaccessibility. While previous literature has focused predominantly on the quality and quantity of real data, our work enables the measurement of diversity in synthetic data and explores its impact on LLM performance. We study the downstream effects of synthetic data diversity during both the pre-training and fine-tuning stages by introducing a new diversity metric, \\textit{LLM cluster-agent}, designed"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.15226","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-19T22:14:07Z","cross_cats_sorted":[],"title_canon_sha256":"78863645172939485dc8e15de79570d2385d76a5ffdb7ffa667d1c24f5f6de57","abstract_canon_sha256":"c371720ba590a69cdf1e52c00208198c54f6bf8923eb1db3c5d13db09e3db338"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:24:18.853405Z","signature_b64":"68sDNvQYL0u+lNnhcnRi7v89t27KSGQ9yx7PVdtvBRFFAI7DenBa88ISNYwj+qvmPZOdlXvf7+8CjrQL5CZcAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"33f8bd65a6c60141d8bf7c7d19c5db76e434ca2c3b957990119d549d00ce344d","last_reissued_at":"2026-07-05T09:24:18.852910Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:24:18.852910Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On the Diversity of Synthetic Data and its Impact on Training Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Abdul Waheed, Bhiksha Raj, Hao Chen, Jindong Wang, Marah I. Abdin, Xiang Li, Yidong Wang","submitted_at":"2024-10-19T22:14:07Z","abstract_excerpt":"The rise of Large Language Models (LLMs) has accentuated the need for diverse, high-quality pre-training data. Synthetic data emerges as a viable solution to the challenges of data scarcity and inaccessibility. While previous literature has focused predominantly on the quality and quantity of real data, our work enables the measurement of diversity in synthetic data and explores its impact on LLM performance. We study the downstream effects of synthetic data diversity during both the pre-training and fine-tuning stages by introducing a new diversity metric, \\textit{LLM cluster-agent}, designed"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.15226","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.15226/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.15226","created_at":"2026-07-05T09:24:18.852969+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.15226v2","created_at":"2026-07-05T09:24:18.852969+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.15226","created_at":"2026-07-05T09:24:18.852969+00:00"},{"alias_kind":"pith_short_12","alias_value":"GP4L2ZNGYYAU","created_at":"2026-07-05T09:24:18.852969+00:00"},{"alias_kind":"pith_short_16","alias_value":"GP4L2ZNGYYAUDWF7","created_at":"2026-07-05T09:24:18.852969+00:00"},{"alias_kind":"pith_short_8","alias_value":"GP4L2ZNG","created_at":"2026-07-05T09:24:18.852969+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2503.08223","citing_title":"Will LLMs Scaling Hit the Wall? Breaking Barriers via Distributed Resources on Massive Edge Devices","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2504.01919","citing_title":"Bridging the Linguistic Divide: A Survey on Leveraging Large Language Models for Machine Translation","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2511.01490","citing_title":"Synthetic Eggs in Many Baskets: The Impact of Synthetic Data Diversity on LLM Fine-Tuning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2512.05929","citing_title":"LLM Harms: A Taxonomy and Discussion","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11922","citing_title":"StepCodeReasoner: Aligning Code Reasoning with Stepwise Execution Traces via Reinforcement Learning","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11290","citing_title":"Polyglot Teachers: Evaluating Language Models for Multilingual Synthetic Data Generation","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GP4L2ZNGYYAUDWF7PR6RTRO3O3","json":"https://pith.science/pith/GP4L2ZNGYYAUDWF7PR6RTRO3O3.json","graph_json":"https://pith.science/api/pith-number/GP4L2ZNGYYAUDWF7PR6RTRO3O3/graph.json","events_json":"https://pith.science/api/pith-number/GP4L2ZNGYYAUDWF7PR6RTRO3O3/events.json","paper":"https://pith.science/paper/GP4L2ZNG"},"agent_actions":{"view_html":"https://pith.science/pith/GP4L2ZNGYYAUDWF7PR6RTRO3O3","download_json":"https://pith.science/pith/GP4L2ZNGYYAUDWF7PR6RTRO3O3.json","view_paper":"https://pith.science/paper/GP4L2ZNG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.15226&json=true","fetch_graph":"https://pith.science/api/pith-number/GP4L2ZNGYYAUDWF7PR6RTRO3O3/graph.json","fetch_events":"https://pith.science/api/pith-number/GP4L2ZNGYYAUDWF7PR6RTRO3O3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GP4L2ZNGYYAUDWF7PR6RTRO3O3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GP4L2ZNGYYAUDWF7PR6RTRO3O3/action/storage_attestation","attest_author":"https://pith.science/pith/GP4L2ZNGYYAUDWF7PR6RTRO3O3/action/author_attestation","sign_citation":"https://pith.science/pith/GP4L2ZNGYYAUDWF7PR6RTRO3O3/action/citation_signature","submit_replication":"https://pith.science/pith/GP4L2ZNGYYAUDWF7PR6RTRO3O3/action/replication_record"}},"created_at":"2026-07-05T09:24:18.852969+00:00","updated_at":"2026-07-05T09:24:18.852969+00:00"}