{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2018:PMX6ERAFWTYHOC5NEIOHCLIO6O","short_pith_number":"pith:PMX6ERAF","schema_version":"1.0","canonical_sha256":"7b2fe24405b4f0770bad221c712d0ef3850afd4379501bca497bdc06b6f1c891","source":{"kind":"arxiv","id":"1805.02109","version":2},"attestation_state":"computed","paper":{"title":"Predicting Race and Ethnicity From the Sequence of Characters in a Name","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"stat.AP","authors_text":"Gaurav Sood, Rajashekar Chintalapati, Suriyan Laohaprapanon","submitted_at":"2018-05-05T20:04:49Z","abstract_excerpt":"To answer questions about racial inequality and fairness, we often need a way to infer race and ethnicity from names. One way to infer race and ethnicity from names is by relying on the Census Bureau's list of popular last names. The list, however, suffers from at least three limitations: 1. it only contains last names, 2. it only includes popular last names, and 3. it is updated once every 10 years. To provide better generalization, and higher accuracy when first names are available, we model the relationship between characters in a name and race and ethnicity using various techniques. A mode"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1805.02109","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.AP","submitted_at":"2018-05-05T20:04:49Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"2135daabaf6442c6bd950b78996184896ee9747248b2b5c6e9f1bd010dd02805","abstract_canon_sha256":"f774c5f8360b2a406f7071ca65fa3306ce9f28083b4d8dcb3320de5013dc9456"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:29:59.007169Z","signature_b64":"xODSjUslRaDfDS4ebRVIY6vmPrd9zib9vCfVq2oPymCdhdeAPB92Gmr3cicsjUjERRHODCmJ9XAya3GfkmFpDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7b2fe24405b4f0770bad221c712d0ef3850afd4379501bca497bdc06b6f1c891","last_reissued_at":"2026-07-05T06:29:59.006671Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:29:59.006671Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Predicting Race and Ethnicity From the Sequence of Characters in a Name","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"stat.AP","authors_text":"Gaurav Sood, Rajashekar Chintalapati, Suriyan Laohaprapanon","submitted_at":"2018-05-05T20:04:49Z","abstract_excerpt":"To answer questions about racial inequality and fairness, we often need a way to infer race and ethnicity from names. One way to infer race and ethnicity from names is by relying on the Census Bureau's list of popular last names. The list, however, suffers from at least three limitations: 1. it only contains last names, 2. it only includes popular last names, and 3. it is updated once every 10 years. To provide better generalization, and higher accuracy when first names are available, we model the relationship between characters in a name and race and ethnicity using various techniques. A mode"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1805.02109","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1805.02109/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1805.02109","created_at":"2026-07-05T06:29:59.006747+00:00"},{"alias_kind":"arxiv_version","alias_value":"1805.02109v2","created_at":"2026-07-05T06:29:59.006747+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1805.02109","created_at":"2026-07-05T06:29:59.006747+00:00"},{"alias_kind":"pith_short_12","alias_value":"PMX6ERAFWTYH","created_at":"2026-07-05T06:29:59.006747+00:00"},{"alias_kind":"pith_short_16","alias_value":"PMX6ERAFWTYHOC5N","created_at":"2026-07-05T06:29:59.006747+00:00"},{"alias_kind":"pith_short_8","alias_value":"PMX6ERAF","created_at":"2026-07-05T06:29:59.006747+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23462","citing_title":"War in the Abstract: The Rise and Consequences of Militarized Language in Scientific Communication","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11091","citing_title":"QUIET: Quantifying Underutilized Influential Edges for Targeted Synchronization","ref_index":105,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28187","citing_title":"Whose Name Comes Up? III: Persona Prompting Effects in LLM-Based Scholar Recommendation","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22555","citing_title":"Using Embedding Models to Improve Probabilistic Race Prediction","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10401","citing_title":"NameBERT: Scaling Name-Based Nationality Classification with LLM-Augmented Open Academic Data","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PMX6ERAFWTYHOC5NEIOHCLIO6O","json":"https://pith.science/pith/PMX6ERAFWTYHOC5NEIOHCLIO6O.json","graph_json":"https://pith.science/api/pith-number/PMX6ERAFWTYHOC5NEIOHCLIO6O/graph.json","events_json":"https://pith.science/api/pith-number/PMX6ERAFWTYHOC5NEIOHCLIO6O/events.json","paper":"https://pith.science/paper/PMX6ERAF"},"agent_actions":{"view_html":"https://pith.science/pith/PMX6ERAFWTYHOC5NEIOHCLIO6O","download_json":"https://pith.science/pith/PMX6ERAFWTYHOC5NEIOHCLIO6O.json","view_paper":"https://pith.science/paper/PMX6ERAF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1805.02109&json=true","fetch_graph":"https://pith.science/api/pith-number/PMX6ERAFWTYHOC5NEIOHCLIO6O/graph.json","fetch_events":"https://pith.science/api/pith-number/PMX6ERAFWTYHOC5NEIOHCLIO6O/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PMX6ERAFWTYHOC5NEIOHCLIO6O/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PMX6ERAFWTYHOC5NEIOHCLIO6O/action/storage_attestation","attest_author":"https://pith.science/pith/PMX6ERAFWTYHOC5NEIOHCLIO6O/action/author_attestation","sign_citation":"https://pith.science/pith/PMX6ERAFWTYHOC5NEIOHCLIO6O/action/citation_signature","submit_replication":"https://pith.science/pith/PMX6ERAFWTYHOC5NEIOHCLIO6O/action/replication_record"}},"created_at":"2026-07-05T06:29:59.006747+00:00","updated_at":"2026-07-05T06:29:59.006747+00:00"}