{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:GT6O6KMUIZ2AUYBVJHD4LLD6XI","short_pith_number":"pith:GT6O6KMU","schema_version":"1.0","canonical_sha256":"34fcef299446740a603549c7c5ac7eba3df7b1926fcb72a4781d633b8ab0221a","source":{"kind":"arxiv","id":"2110.00672","version":1},"attestation_state":"computed","paper":{"title":"Low Frequency Names Exhibit Bias and Overfitting in Contextualizing Language Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CY","authors_text":"Aylin Caliskan, Robert Wolfe","submitted_at":"2021-10-01T22:44:31Z","abstract_excerpt":"We use a dataset of U.S. first names with labels based on predominant gender and racial group to examine the effect of training corpus frequency on tokenization, contextualization, similarity to initial representation, and bias in BERT, GPT-2, T5, and XLNet. We show that predominantly female and non-white names are less frequent in the training corpora of these four language models. We find that infrequent names are more self-similar across contexts, with Spearman's r between frequency and self-similarity as low as -.763. Infrequent names are also less similar to initial representation, with S"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2110.00672","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CY","submitted_at":"2021-10-01T22:44:31Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"c9a054c05812ca0b34cc2bb339686b8d43cca2f4c8de558d5e4c8d543a5ef6b8","abstract_canon_sha256":"dde1ad2e9fb7f46ea88506bdd21b777b3c6b85e58522958813d261b17cec208e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:19:47.001588Z","signature_b64":"y4y8EWZxLJDPco5pGk3Uux85mYO7XkAidQe+GLEwqcHMIxOpHt4VNiviPG7DIpg0zzbwC+6FJ7T1IJyBjEucCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"34fcef299446740a603549c7c5ac7eba3df7b1926fcb72a4781d633b8ab0221a","last_reissued_at":"2026-07-05T03:19:47.001204Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:19:47.001204Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Low Frequency Names Exhibit Bias and Overfitting in Contextualizing Language Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CY","authors_text":"Aylin Caliskan, Robert Wolfe","submitted_at":"2021-10-01T22:44:31Z","abstract_excerpt":"We use a dataset of U.S. first names with labels based on predominant gender and racial group to examine the effect of training corpus frequency on tokenization, contextualization, similarity to initial representation, and bias in BERT, GPT-2, T5, and XLNet. We show that predominantly female and non-white names are less frequent in the training corpora of these four language models. We find that infrequent names are more self-similar across contexts, with Spearman's r between frequency and self-similarity as low as -.763. Infrequent names are also less similar to initial representation, with S"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2110.00672","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2110.00672/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2110.00672","created_at":"2026-07-05T03:19:47.001259+00:00"},{"alias_kind":"arxiv_version","alias_value":"2110.00672v1","created_at":"2026-07-05T03:19:47.001259+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2110.00672","created_at":"2026-07-05T03:19:47.001259+00:00"},{"alias_kind":"pith_short_12","alias_value":"GT6O6KMUIZ2A","created_at":"2026-07-05T03:19:47.001259+00:00"},{"alias_kind":"pith_short_16","alias_value":"GT6O6KMUIZ2AUYBV","created_at":"2026-07-05T03:19:47.001259+00:00"},{"alias_kind":"pith_short_8","alias_value":"GT6O6KMU","created_at":"2026-07-05T03:19:47.001259+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GT6O6KMUIZ2AUYBVJHD4LLD6XI","json":"https://pith.science/pith/GT6O6KMUIZ2AUYBVJHD4LLD6XI.json","graph_json":"https://pith.science/api/pith-number/GT6O6KMUIZ2AUYBVJHD4LLD6XI/graph.json","events_json":"https://pith.science/api/pith-number/GT6O6KMUIZ2AUYBVJHD4LLD6XI/events.json","paper":"https://pith.science/paper/GT6O6KMU"},"agent_actions":{"view_html":"https://pith.science/pith/GT6O6KMUIZ2AUYBVJHD4LLD6XI","download_json":"https://pith.science/pith/GT6O6KMUIZ2AUYBVJHD4LLD6XI.json","view_paper":"https://pith.science/paper/GT6O6KMU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2110.00672&json=true","fetch_graph":"https://pith.science/api/pith-number/GT6O6KMUIZ2AUYBVJHD4LLD6XI/graph.json","fetch_events":"https://pith.science/api/pith-number/GT6O6KMUIZ2AUYBVJHD4LLD6XI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GT6O6KMUIZ2AUYBVJHD4LLD6XI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GT6O6KMUIZ2AUYBVJHD4LLD6XI/action/storage_attestation","attest_author":"https://pith.science/pith/GT6O6KMUIZ2AUYBVJHD4LLD6XI/action/author_attestation","sign_citation":"https://pith.science/pith/GT6O6KMUIZ2AUYBVJHD4LLD6XI/action/citation_signature","submit_replication":"https://pith.science/pith/GT6O6KMUIZ2AUYBVJHD4LLD6XI/action/replication_record"}},"created_at":"2026-07-05T03:19:47.001259+00:00","updated_at":"2026-07-05T03:19:47.001259+00:00"}