{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:5IABA4LE7QQXYJKXXPIUB33FQ7","short_pith_number":"pith:5IABA4LE","schema_version":"1.0","canonical_sha256":"ea00107164fc217c2557bbd140ef6587d0ed5d187691063b0bbab816b458f359","source":{"kind":"arxiv","id":"2307.04626","version":2},"attestation_state":"computed","paper":{"title":"Measuring Lexical Diversity in Texts: The Twofold Length Problem","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["stat.AP"],"primary_cat":"cs.CL","authors_text":"Yves Bestgen","submitted_at":"2023-07-10T15:10:56Z","abstract_excerpt":"The impact of text length on the estimation of lexical diversity has captured the attention of the scientific community for more than a century. Numerous indices have been proposed, and many studies have been conducted to evaluate them, but the problem remains. This methodological review provides a critical analysis not only of the most commonly used indices in language learning studies, but also of the length problem itself, as well as of the methodology for evaluating the proposed solutions. The analysis of three datasets of English language-learners' texts revealed that indices that reduce "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.04626","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2023-07-10T15:10:56Z","cross_cats_sorted":["stat.AP"],"title_canon_sha256":"3a9044ecc5d9a1edc00916284c8ef08b7c0e692c56ca792e559cbf6112d2bed0","abstract_canon_sha256":"88680574d82bb0520f4486b4335836c82bead1a495277134bd507153aa01f8f7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:35:52.824789Z","signature_b64":"snKrEqLl1cMjpjMKJ15qKjM88s/tiB3Tbj9nNJdgmrTGLB2bv+p7F2L7F3iMtankehFG62aHKbpk+kn6hywYBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ea00107164fc217c2557bbd140ef6587d0ed5d187691063b0bbab816b458f359","last_reissued_at":"2026-07-05T06:35:52.824374Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:35:52.824374Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Measuring Lexical Diversity in Texts: The Twofold Length Problem","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["stat.AP"],"primary_cat":"cs.CL","authors_text":"Yves Bestgen","submitted_at":"2023-07-10T15:10:56Z","abstract_excerpt":"The impact of text length on the estimation of lexical diversity has captured the attention of the scientific community for more than a century. Numerous indices have been proposed, and many studies have been conducted to evaluate them, but the problem remains. This methodological review provides a critical analysis not only of the most commonly used indices in language learning studies, but also of the length problem itself, as well as of the methodology for evaluating the proposed solutions. The analysis of three datasets of English language-learners' texts revealed that indices that reduce "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.04626","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.04626/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.04626","created_at":"2026-07-05T06:35:52.824432+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.04626v2","created_at":"2026-07-05T06:35:52.824432+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.04626","created_at":"2026-07-05T06:35:52.824432+00:00"},{"alias_kind":"pith_short_12","alias_value":"5IABA4LE7QQX","created_at":"2026-07-05T06:35:52.824432+00:00"},{"alias_kind":"pith_short_16","alias_value":"5IABA4LE7QQXYJKX","created_at":"2026-07-05T06:35:52.824432+00:00"},{"alias_kind":"pith_short_8","alias_value":"5IABA4LE","created_at":"2026-07-05T06:35:52.824432+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.08014","citing_title":"Mass-Scale Analysis of In-the-Wild Conversations Reveals Complexity Bounds on LLM Jailbreaking","ref_index":32,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5IABA4LE7QQXYJKXXPIUB33FQ7","json":"https://pith.science/pith/5IABA4LE7QQXYJKXXPIUB33FQ7.json","graph_json":"https://pith.science/api/pith-number/5IABA4LE7QQXYJKXXPIUB33FQ7/graph.json","events_json":"https://pith.science/api/pith-number/5IABA4LE7QQXYJKXXPIUB33FQ7/events.json","paper":"https://pith.science/paper/5IABA4LE"},"agent_actions":{"view_html":"https://pith.science/pith/5IABA4LE7QQXYJKXXPIUB33FQ7","download_json":"https://pith.science/pith/5IABA4LE7QQXYJKXXPIUB33FQ7.json","view_paper":"https://pith.science/paper/5IABA4LE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.04626&json=true","fetch_graph":"https://pith.science/api/pith-number/5IABA4LE7QQXYJKXXPIUB33FQ7/graph.json","fetch_events":"https://pith.science/api/pith-number/5IABA4LE7QQXYJKXXPIUB33FQ7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5IABA4LE7QQXYJKXXPIUB33FQ7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5IABA4LE7QQXYJKXXPIUB33FQ7/action/storage_attestation","attest_author":"https://pith.science/pith/5IABA4LE7QQXYJKXXPIUB33FQ7/action/author_attestation","sign_citation":"https://pith.science/pith/5IABA4LE7QQXYJKXXPIUB33FQ7/action/citation_signature","submit_replication":"https://pith.science/pith/5IABA4LE7QQXYJKXXPIUB33FQ7/action/replication_record"}},"created_at":"2026-07-05T06:35:52.824432+00:00","updated_at":"2026-07-05T06:35:52.824432+00:00"}