{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:EMHZVHI2QMCZDAILEXGGCDRLXC","short_pith_number":"pith:EMHZVHI2","schema_version":"1.0","canonical_sha256":"230f9a9d1a830591810b25cc610e2bb8bd123897e960f3ef5a4df3d0e2e2fe56","source":{"kind":"arxiv","id":"2607.17043","version":1},"attestation_state":"computed","paper":{"title":"Learning from Synthetic Data without Model Collapse in Iterative Instruction Tuning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chuan Zou, Kehan Guo, Ping He, Ting Hua, Xiangliang Zhang, Xiaonan Luo, Yue Huang","submitted_at":"2026-07-19T03:20:01Z","abstract_excerpt":"Model collapse is a central challenge in learning from synthetic data: as later-generation large language models (LLMs) are trained on an increasing proportion of model-generated data, performance can degrade due to narrowed coverage and accumulated bias. Existing work mainly studies how to bound this degradation. In iterative model evolution, however, the more meaningful objective is to ensure that each successive model improves over its predecessor, which requires diagnosing collapse at a granularity that is actionable for data curation. We study this problem in synthetic data self-improving"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.17043","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2026-07-19T03:20:01Z","cross_cats_sorted":[],"title_canon_sha256":"546b922cddac346f592e58a6f6afbbf932169f74e645c27fc10828a91d77096f","abstract_canon_sha256":"10d05009af3fe657c10fde2f62120ce63266bf3785772ce3ce1a036a5a385ffc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-21T01:21:12.019717Z","signature_b64":"QYr1u++rV8wRgoSImgF/KXer2yppN1bi7ArcC5h75vxWl7IN/ht7BgaezWSv+AVc4R4a9KxjdX2K+9QhrLtKBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"230f9a9d1a830591810b25cc610e2bb8bd123897e960f3ef5a4df3d0e2e2fe56","last_reissued_at":"2026-07-21T01:21:12.018867Z","signature_status":"signed_v1","first_computed_at":"2026-07-21T01:21:12.018867Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning from Synthetic Data without Model Collapse in Iterative Instruction Tuning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chuan Zou, Kehan Guo, Ping He, Ting Hua, Xiangliang Zhang, Xiaonan Luo, Yue Huang","submitted_at":"2026-07-19T03:20:01Z","abstract_excerpt":"Model collapse is a central challenge in learning from synthetic data: as later-generation large language models (LLMs) are trained on an increasing proportion of model-generated data, performance can degrade due to narrowed coverage and accumulated bias. Existing work mainly studies how to bound this degradation. In iterative model evolution, however, the more meaningful objective is to ensure that each successive model improves over its predecessor, which requires diagnosing collapse at a granularity that is actionable for data curation. We study this problem in synthetic data self-improving"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.17043","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.17043/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.17043","created_at":"2026-07-21T01:21:12.019306+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.17043v1","created_at":"2026-07-21T01:21:12.019306+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.17043","created_at":"2026-07-21T01:21:12.019306+00:00"},{"alias_kind":"pith_short_12","alias_value":"EMHZVHI2QMCZ","created_at":"2026-07-21T01:21:12.019306+00:00"},{"alias_kind":"pith_short_16","alias_value":"EMHZVHI2QMCZDAIL","created_at":"2026-07-21T01:21:12.019306+00:00"},{"alias_kind":"pith_short_8","alias_value":"EMHZVHI2","created_at":"2026-07-21T01:21:12.019306+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.01522","citing_title":"Question Begets Question: Self-Evolving Curriculum for Reinforcement Fine-Tuning on Competition Mathematics","ref_index":28,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EMHZVHI2QMCZDAILEXGGCDRLXC","json":"https://pith.science/pith/EMHZVHI2QMCZDAILEXGGCDRLXC.json","graph_json":"https://pith.science/api/pith-number/EMHZVHI2QMCZDAILEXGGCDRLXC/graph.json","events_json":"https://pith.science/api/pith-number/EMHZVHI2QMCZDAILEXGGCDRLXC/events.json","paper":"https://pith.science/paper/EMHZVHI2"},"agent_actions":{"view_html":"https://pith.science/pith/EMHZVHI2QMCZDAILEXGGCDRLXC","download_json":"https://pith.science/pith/EMHZVHI2QMCZDAILEXGGCDRLXC.json","view_paper":"https://pith.science/paper/EMHZVHI2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.17043&json=true","fetch_graph":"https://pith.science/api/pith-number/EMHZVHI2QMCZDAILEXGGCDRLXC/graph.json","fetch_events":"https://pith.science/api/pith-number/EMHZVHI2QMCZDAILEXGGCDRLXC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EMHZVHI2QMCZDAILEXGGCDRLXC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EMHZVHI2QMCZDAILEXGGCDRLXC/action/storage_attestation","attest_author":"https://pith.science/pith/EMHZVHI2QMCZDAILEXGGCDRLXC/action/author_attestation","sign_citation":"https://pith.science/pith/EMHZVHI2QMCZDAILEXGGCDRLXC/action/citation_signature","submit_replication":"https://pith.science/pith/EMHZVHI2QMCZDAILEXGGCDRLXC/action/replication_record"}},"created_at":"2026-07-21T01:21:12.019306+00:00","updated_at":"2026-07-21T01:21:12.019306+00:00"}