{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:SMGYSLQ5OTLUYNYNMPRLBITF7L","short_pith_number":"pith:SMGYSLQ5","schema_version":"1.0","canonical_sha256":"930d892e1d74d74c370d63e2b0a265fad4a7214c37b34189187d7992b5248914","source":{"kind":"arxiv","id":"2106.02289","version":1},"attestation_state":"computed","paper":{"title":"Modeling the Unigram Distribution","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dami\\'an E. Blasi, Irene Nikkarinen, Ryan Cotterell, Tiago Pimentel","submitted_at":"2021-06-04T07:02:49Z","abstract_excerpt":"The unigram distribution is the non-contextual probability of finding a specific word form in a corpus. While of central importance to the study of language, it is commonly approximated by each word's sample frequency in the corpus. This approach, being highly dependent on sample size, assigns zero probability to any out-of-vocabulary (oov) word form. As a result, it produces negatively biased probabilities for any oov word form, while positively biased probabilities to in-corpus words. In this work, we argue in favor of properly modeling the unigram distribution -- claiming it should be a cen"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.02289","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-06-04T07:02:49Z","cross_cats_sorted":[],"title_canon_sha256":"59897e662a5ad44916c7255166ff00a9ecb38ec4852bd2f7848142af7a3bd759","abstract_canon_sha256":"abc5c6a35212719eed78fd13a36d754732b5fc553c63671cdf49704686b49290"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:46:17.760748Z","signature_b64":"A3wAZqW/bdno7ZyJmqGWNZhkoLPPlaiwMebptQF2y14br6MzC593mhtgPAqC4eDoo2Zq3e+OB69iVBRDhD18Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"930d892e1d74d74c370d63e2b0a265fad4a7214c37b34189187d7992b5248914","last_reissued_at":"2026-07-05T02:46:17.760260Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:46:17.760260Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Modeling the Unigram Distribution","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dami\\'an E. Blasi, Irene Nikkarinen, Ryan Cotterell, Tiago Pimentel","submitted_at":"2021-06-04T07:02:49Z","abstract_excerpt":"The unigram distribution is the non-contextual probability of finding a specific word form in a corpus. While of central importance to the study of language, it is commonly approximated by each word's sample frequency in the corpus. This approach, being highly dependent on sample size, assigns zero probability to any out-of-vocabulary (oov) word form. As a result, it produces negatively biased probabilities for any oov word form, while positively biased probabilities to in-corpus words. In this work, we argue in favor of properly modeling the unigram distribution -- claiming it should be a cen"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.02289","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.02289/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.02289","created_at":"2026-07-05T02:46:17.760320+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.02289v1","created_at":"2026-07-05T02:46:17.760320+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.02289","created_at":"2026-07-05T02:46:17.760320+00:00"},{"alias_kind":"pith_short_12","alias_value":"SMGYSLQ5OTLU","created_at":"2026-07-05T02:46:17.760320+00:00"},{"alias_kind":"pith_short_16","alias_value":"SMGYSLQ5OTLUYNYN","created_at":"2026-07-05T02:46:17.760320+00:00"},{"alias_kind":"pith_short_8","alias_value":"SMGYSLQ5","created_at":"2026-07-05T02:46:17.760320+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.25634","citing_title":"The Surprising Universality of LLM Outputs: A Real-Time Verification Primitive","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SMGYSLQ5OTLUYNYNMPRLBITF7L","json":"https://pith.science/pith/SMGYSLQ5OTLUYNYNMPRLBITF7L.json","graph_json":"https://pith.science/api/pith-number/SMGYSLQ5OTLUYNYNMPRLBITF7L/graph.json","events_json":"https://pith.science/api/pith-number/SMGYSLQ5OTLUYNYNMPRLBITF7L/events.json","paper":"https://pith.science/paper/SMGYSLQ5"},"agent_actions":{"view_html":"https://pith.science/pith/SMGYSLQ5OTLUYNYNMPRLBITF7L","download_json":"https://pith.science/pith/SMGYSLQ5OTLUYNYNMPRLBITF7L.json","view_paper":"https://pith.science/paper/SMGYSLQ5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.02289&json=true","fetch_graph":"https://pith.science/api/pith-number/SMGYSLQ5OTLUYNYNMPRLBITF7L/graph.json","fetch_events":"https://pith.science/api/pith-number/SMGYSLQ5OTLUYNYNMPRLBITF7L/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SMGYSLQ5OTLUYNYNMPRLBITF7L/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SMGYSLQ5OTLUYNYNMPRLBITF7L/action/storage_attestation","attest_author":"https://pith.science/pith/SMGYSLQ5OTLUYNYNMPRLBITF7L/action/author_attestation","sign_citation":"https://pith.science/pith/SMGYSLQ5OTLUYNYNMPRLBITF7L/action/citation_signature","submit_replication":"https://pith.science/pith/SMGYSLQ5OTLUYNYNMPRLBITF7L/action/replication_record"}},"created_at":"2026-07-05T02:46:17.760320+00:00","updated_at":"2026-07-05T02:46:17.760320+00:00"}