{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:F77MT2DKH7EAEUTGWMSXBSJ3P5","short_pith_number":"pith:F77MT2DK","schema_version":"1.0","canonical_sha256":"2ffec9e86a3fc8025266b32570c93b7f59fbb4e3de89596d180ab7f2c961a62e","source":{"kind":"arxiv","id":"2003.08529","version":1},"attestation_state":"computed","paper":{"title":"Diversity, Density, and Homogeneity: Quantitative Characteristic Metrics for Text Collections","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Mona Diab, Xuan Zhu, Yi-An Lai, Yi Zhang","submitted_at":"2020-03-19T00:48:32Z","abstract_excerpt":"Summarizing data samples by quantitative measures has a long history, with descriptive statistics being a case in point. However, as natural language processing methods flourish, there are still insufficient characteristic metrics to describe a collection of texts in terms of the words, sentences, or paragraphs they comprise. In this work, we propose metrics of diversity, density, and homogeneity that quantitatively measure the dispersion, sparsity, and uniformity of a text collection. We conduct a series of simulations to verify that each metric holds desired properties and resonates with hum"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2003.08529","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-03-19T00:48:32Z","cross_cats_sorted":[],"title_canon_sha256":"56927921044ba119d75a9d7414692d0955b8245567745a063324a64451a67d12","abstract_canon_sha256":"c690bbee7297fcea97b74097c4a430c54b73582d79ac2331ea6f9a9742f1f3f7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:49:11.551571Z","signature_b64":"nyWcfXCSqeZbPeXVX3nCWLmADYjsBdJow3GVnnJwnFL421fVuQRznc73L579A0CsTEPqEcDJvWpAoyu9JIMkBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2ffec9e86a3fc8025266b32570c93b7f59fbb4e3de89596d180ab7f2c961a62e","last_reissued_at":"2026-07-05T00:49:11.551125Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:49:11.551125Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Diversity, Density, and Homogeneity: Quantitative Characteristic Metrics for Text Collections","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Mona Diab, Xuan Zhu, Yi-An Lai, Yi Zhang","submitted_at":"2020-03-19T00:48:32Z","abstract_excerpt":"Summarizing data samples by quantitative measures has a long history, with descriptive statistics being a case in point. However, as natural language processing methods flourish, there are still insufficient characteristic metrics to describe a collection of texts in terms of the words, sentences, or paragraphs they comprise. In this work, we propose metrics of diversity, density, and homogeneity that quantitatively measure the dispersion, sparsity, and uniformity of a text collection. We conduct a series of simulations to verify that each metric holds desired properties and resonates with hum"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2003.08529","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2003.08529/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2003.08529","created_at":"2026-07-05T00:49:11.551176+00:00"},{"alias_kind":"arxiv_version","alias_value":"2003.08529v1","created_at":"2026-07-05T00:49:11.551176+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2003.08529","created_at":"2026-07-05T00:49:11.551176+00:00"},{"alias_kind":"pith_short_12","alias_value":"F77MT2DKH7EA","created_at":"2026-07-05T00:49:11.551176+00:00"},{"alias_kind":"pith_short_16","alias_value":"F77MT2DKH7EAEUTG","created_at":"2026-07-05T00:49:11.551176+00:00"},{"alias_kind":"pith_short_8","alias_value":"F77MT2DK","created_at":"2026-07-05T00:49:11.551176+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.08512","citing_title":"Measuring Diversity in Synthetic Datasets","ref_index":35,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F77MT2DKH7EAEUTGWMSXBSJ3P5","json":"https://pith.science/pith/F77MT2DKH7EAEUTGWMSXBSJ3P5.json","graph_json":"https://pith.science/api/pith-number/F77MT2DKH7EAEUTGWMSXBSJ3P5/graph.json","events_json":"https://pith.science/api/pith-number/F77MT2DKH7EAEUTGWMSXBSJ3P5/events.json","paper":"https://pith.science/paper/F77MT2DK"},"agent_actions":{"view_html":"https://pith.science/pith/F77MT2DKH7EAEUTGWMSXBSJ3P5","download_json":"https://pith.science/pith/F77MT2DKH7EAEUTGWMSXBSJ3P5.json","view_paper":"https://pith.science/paper/F77MT2DK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2003.08529&json=true","fetch_graph":"https://pith.science/api/pith-number/F77MT2DKH7EAEUTGWMSXBSJ3P5/graph.json","fetch_events":"https://pith.science/api/pith-number/F77MT2DKH7EAEUTGWMSXBSJ3P5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F77MT2DKH7EAEUTGWMSXBSJ3P5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F77MT2DKH7EAEUTGWMSXBSJ3P5/action/storage_attestation","attest_author":"https://pith.science/pith/F77MT2DKH7EAEUTGWMSXBSJ3P5/action/author_attestation","sign_citation":"https://pith.science/pith/F77MT2DKH7EAEUTGWMSXBSJ3P5/action/citation_signature","submit_replication":"https://pith.science/pith/F77MT2DKH7EAEUTGWMSXBSJ3P5/action/replication_record"}},"created_at":"2026-07-05T00:49:11.551176+00:00","updated_at":"2026-07-05T00:49:11.551176+00:00"}