{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:PQWRX2JK6CVKF4BHS2POKAQOBG","short_pith_number":"pith:PQWRX2JK","schema_version":"1.0","canonical_sha256":"7c2d1be92af0aaa2f027969ee5020e098218480fa14508febef3afb768b8b4c8","source":{"kind":"arxiv","id":"2311.09807","version":2},"attestation_state":"computed","paper":{"title":"The Curious Decline of Linguistic Diversity: Training Language Models on Synthetic Text","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chlo\\'e Clavel, Guokan Shang, Michalis Vazirgiannis, Yanzhu Guo","submitted_at":"2023-11-16T11:31:50Z","abstract_excerpt":"This study investigates the consequences of training language models on synthetic data generated by their predecessors, an increasingly prevalent practice given the prominence of powerful generative models. Diverging from the usual emphasis on performance metrics, we focus on the impact of this training methodology on linguistic diversity, especially when conducted recursively over time. To assess this, we adapt and develop a set of novel metrics targeting lexical, syntactic, and semantic diversity, applying them in recursive finetuning experiments across various natural language generation ta"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.09807","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-11-16T11:31:50Z","cross_cats_sorted":[],"title_canon_sha256":"0381f9b9ecb9b241ef8bd147d4100532afd8584e7c3b86170c90cbaefc3364b0","abstract_canon_sha256":"5ee82796b07a8fd93a18a337769fdb9198493efb7b9982e2aaf858379bf59f07"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:08:40.254979Z","signature_b64":"whNr6dxHKnOMZo3uBvLH54K1YNrkms1bFqSi9UB/jZmFWLbpFKoQOdgn71bQy3R62+Sh3FQP7h/v5/FqUJnkBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7c2d1be92af0aaa2f027969ee5020e098218480fa14508febef3afb768b8b4c8","last_reissued_at":"2026-07-05T08:08:40.254500Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:08:40.254500Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Curious Decline of Linguistic Diversity: Training Language Models on Synthetic Text","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chlo\\'e Clavel, Guokan Shang, Michalis Vazirgiannis, Yanzhu Guo","submitted_at":"2023-11-16T11:31:50Z","abstract_excerpt":"This study investigates the consequences of training language models on synthetic data generated by their predecessors, an increasingly prevalent practice given the prominence of powerful generative models. Diverging from the usual emphasis on performance metrics, we focus on the impact of this training methodology on linguistic diversity, especially when conducted recursively over time. To assess this, we adapt and develop a set of novel metrics targeting lexical, syntactic, and semantic diversity, applying them in recursive finetuning experiments across various natural language generation ta"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.09807","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.09807/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.09807","created_at":"2026-07-05T08:08:40.254555+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.09807v2","created_at":"2026-07-05T08:08:40.254555+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.09807","created_at":"2026-07-05T08:08:40.254555+00:00"},{"alias_kind":"pith_short_12","alias_value":"PQWRX2JK6CVK","created_at":"2026-07-05T08:08:40.254555+00:00"},{"alias_kind":"pith_short_16","alias_value":"PQWRX2JK6CVKF4BH","created_at":"2026-07-05T08:08:40.254555+00:00"},{"alias_kind":"pith_short_8","alias_value":"PQWRX2JK","created_at":"2026-07-05T08:08:40.254555+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.30815","citing_title":"When transformers learn \"impossible\" languages, what do they learn?","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21853","citing_title":"Generative artificial intelligence reduces social welfare through model collapse","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01130","citing_title":"Iterative Finetuning is Mostly Idempotent","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PQWRX2JK6CVKF4BHS2POKAQOBG","json":"https://pith.science/pith/PQWRX2JK6CVKF4BHS2POKAQOBG.json","graph_json":"https://pith.science/api/pith-number/PQWRX2JK6CVKF4BHS2POKAQOBG/graph.json","events_json":"https://pith.science/api/pith-number/PQWRX2JK6CVKF4BHS2POKAQOBG/events.json","paper":"https://pith.science/paper/PQWRX2JK"},"agent_actions":{"view_html":"https://pith.science/pith/PQWRX2JK6CVKF4BHS2POKAQOBG","download_json":"https://pith.science/pith/PQWRX2JK6CVKF4BHS2POKAQOBG.json","view_paper":"https://pith.science/paper/PQWRX2JK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.09807&json=true","fetch_graph":"https://pith.science/api/pith-number/PQWRX2JK6CVKF4BHS2POKAQOBG/graph.json","fetch_events":"https://pith.science/api/pith-number/PQWRX2JK6CVKF4BHS2POKAQOBG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PQWRX2JK6CVKF4BHS2POKAQOBG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PQWRX2JK6CVKF4BHS2POKAQOBG/action/storage_attestation","attest_author":"https://pith.science/pith/PQWRX2JK6CVKF4BHS2POKAQOBG/action/author_attestation","sign_citation":"https://pith.science/pith/PQWRX2JK6CVKF4BHS2POKAQOBG/action/citation_signature","submit_replication":"https://pith.science/pith/PQWRX2JK6CVKF4BHS2POKAQOBG/action/replication_record"}},"created_at":"2026-07-05T08:08:40.254555+00:00","updated_at":"2026-07-05T08:08:40.254555+00:00"}