{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RJM55CES4VIXAJR3UG6MC5TGJY","short_pith_number":"pith:RJM55CES","schema_version":"1.0","canonical_sha256":"8a59de8892e55170263ba1bcc176664e33624f11483d37a577b817beb478263a","source":{"kind":"arxiv","id":"2412.10271","version":2},"attestation_state":"computed","paper":{"title":"Benchmarking Linguistic Diversity of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chlo\\'e Clavel, Guokan Shang, Yanzhu Guo","submitted_at":"2024-12-13T16:46:03Z","abstract_excerpt":"The development and evaluation of Large Language Models (LLMs) has primarily focused on their task-solving capabilities, with recent models even surpassing human performance in some areas. However, this focus often neglects whether machine-generated language matches the human level of diversity, in terms of vocabulary choice, syntactic construction, and expression of meaning, raising questions about whether the fundamentals of language generation have been fully addressed. This paper emphasizes the importance of examining the preservation of human linguistic richness by language models, given "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.10271","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-12-13T16:46:03Z","cross_cats_sorted":[],"title_canon_sha256":"cc3a91fcba21d13df3e969bfb9944d2a7c7d08a27cb91d85adf35c8e01b4e61e","abstract_canon_sha256":"bd7a342ded8eb76d884e8eb6158459f0bb7e9470698a3a8cfcc88e7489d62007"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:43:30.174117Z","signature_b64":"LafZTCCHSfQ7+fjKJ2QeRIzIOoIETofx/qCxE46rJ2vkWuLnB6U8XtQspu0zCv8coyQY9KEHEO0TrM+492/pCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8a59de8892e55170263ba1bcc176664e33624f11483d37a577b817beb478263a","last_reissued_at":"2026-07-05T11:43:30.173577Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:43:30.173577Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Benchmarking Linguistic Diversity of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chlo\\'e Clavel, Guokan Shang, Yanzhu Guo","submitted_at":"2024-12-13T16:46:03Z","abstract_excerpt":"The development and evaluation of Large Language Models (LLMs) has primarily focused on their task-solving capabilities, with recent models even surpassing human performance in some areas. However, this focus often neglects whether machine-generated language matches the human level of diversity, in terms of vocabulary choice, syntactic construction, and expression of meaning, raising questions about whether the fundamentals of language generation have been fully addressed. This paper emphasizes the importance of examining the preservation of human linguistic richness by language models, given "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.10271","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.10271/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.10271","created_at":"2026-07-05T11:43:30.173633+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.10271v2","created_at":"2026-07-05T11:43:30.173633+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.10271","created_at":"2026-07-05T11:43:30.173633+00:00"},{"alias_kind":"pith_short_12","alias_value":"RJM55CES4VIX","created_at":"2026-07-05T11:43:30.173633+00:00"},{"alias_kind":"pith_short_16","alias_value":"RJM55CES4VIXAJR3","created_at":"2026-07-05T11:43:30.173633+00:00"},{"alias_kind":"pith_short_8","alias_value":"RJM55CES","created_at":"2026-07-05T11:43:30.173633+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21645","citing_title":"Behavioral and Representational Evidence of Binomial Ordering Preferences in Large Language Models","ref_index":255,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01811","citing_title":"\"I've Seen How This Goes\": Characterizing Diversity via Progressive Conditional Surprise","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2507.03933","citing_title":"Losing our Tail, Again: (Un)Natural Selection & Multilingual LLMs","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2507.21934","citing_title":"Culinary Crossroads: A RAG Framework for Enhancing Diversity in Cross-Cultural Recipe Adaptation","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27673","citing_title":"The TEA Nets framework combines AI and cognitive network science to model targets, events and actors in text","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06030","citing_title":"More Aligned, Less Diverse? Analyzing the Grammar and Lexicon of Two Generations of LLMs","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RJM55CES4VIXAJR3UG6MC5TGJY","json":"https://pith.science/pith/RJM55CES4VIXAJR3UG6MC5TGJY.json","graph_json":"https://pith.science/api/pith-number/RJM55CES4VIXAJR3UG6MC5TGJY/graph.json","events_json":"https://pith.science/api/pith-number/RJM55CES4VIXAJR3UG6MC5TGJY/events.json","paper":"https://pith.science/paper/RJM55CES"},"agent_actions":{"view_html":"https://pith.science/pith/RJM55CES4VIXAJR3UG6MC5TGJY","download_json":"https://pith.science/pith/RJM55CES4VIXAJR3UG6MC5TGJY.json","view_paper":"https://pith.science/paper/RJM55CES","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.10271&json=true","fetch_graph":"https://pith.science/api/pith-number/RJM55CES4VIXAJR3UG6MC5TGJY/graph.json","fetch_events":"https://pith.science/api/pith-number/RJM55CES4VIXAJR3UG6MC5TGJY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RJM55CES4VIXAJR3UG6MC5TGJY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RJM55CES4VIXAJR3UG6MC5TGJY/action/storage_attestation","attest_author":"https://pith.science/pith/RJM55CES4VIXAJR3UG6MC5TGJY/action/author_attestation","sign_citation":"https://pith.science/pith/RJM55CES4VIXAJR3UG6MC5TGJY/action/citation_signature","submit_replication":"https://pith.science/pith/RJM55CES4VIXAJR3UG6MC5TGJY/action/replication_record"}},"created_at":"2026-07-05T11:43:30.173633+00:00","updated_at":"2026-07-05T11:43:30.173633+00:00"}