{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FTIMO5DTWDURLFGUSBJH7BU4JW","short_pith_number":"pith:FTIMO5DT","schema_version":"1.0","canonical_sha256":"2cd0c77473b0e91594d490527f869c4dabd33f6a8baf182008bff9303124a37b","source":{"kind":"arxiv","id":"2409.11491","version":1},"attestation_state":"computed","paper":{"title":"Enriching Datasets with Demographics through Large Language Models: What's in a Name?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Abdulla AlKetbi, Andreas Henschel, Gautier Marti, Khaled AlNuaimi, Mathieu Ravaut, Raed Jaradat","submitted_at":"2024-09-17T18:40:49Z","abstract_excerpt":"Enriching datasets with demographic information, such as gender, race, and age from names, is a critical task in fields like healthcare, public policy, and social sciences. Such demographic insights allow for more precise and effective engagement with target populations. Despite previous efforts employing hidden Markov models and recurrent neural networks to predict demographics from names, significant limitations persist: the lack of large-scale, well-curated, unbiased, publicly available datasets, and the lack of an approach robust across datasets. This scarcity has hindered the development "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.11491","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-09-17T18:40:49Z","cross_cats_sorted":[],"title_canon_sha256":"492d318464ba669c1848f368d974feca342e5cb308dc8ff0e906ebbd45e5810e","abstract_canon_sha256":"651e9069de915cbc91e81076c29bd330e2f1b2059233d8104426da1f428cc9dc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:08:32.330886Z","signature_b64":"r81VcAzhlb8gkmzjZ4G0P6wlkHx3KJY3bL8r43V/BiuPaFdP1xaIcXCVq14O2y/UhqD2CwbvO/Q3i2tcsIx9BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2cd0c77473b0e91594d490527f869c4dabd33f6a8baf182008bff9303124a37b","last_reissued_at":"2026-07-05T09:08:32.330270Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:08:32.330270Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Enriching Datasets with Demographics through Large Language Models: What's in a Name?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Abdulla AlKetbi, Andreas Henschel, Gautier Marti, Khaled AlNuaimi, Mathieu Ravaut, Raed Jaradat","submitted_at":"2024-09-17T18:40:49Z","abstract_excerpt":"Enriching datasets with demographic information, such as gender, race, and age from names, is a critical task in fields like healthcare, public policy, and social sciences. Such demographic insights allow for more precise and effective engagement with target populations. Despite previous efforts employing hidden Markov models and recurrent neural networks to predict demographics from names, significant limitations persist: the lack of large-scale, well-curated, unbiased, publicly available datasets, and the lack of an approach robust across datasets. This scarcity has hindered the development "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.11491","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.11491/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.11491","created_at":"2026-07-05T09:08:32.330343+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.11491v1","created_at":"2026-07-05T09:08:32.330343+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.11491","created_at":"2026-07-05T09:08:32.330343+00:00"},{"alias_kind":"pith_short_12","alias_value":"FTIMO5DTWDUR","created_at":"2026-07-05T09:08:32.330343+00:00"},{"alias_kind":"pith_short_16","alias_value":"FTIMO5DTWDURLFGU","created_at":"2026-07-05T09:08:32.330343+00:00"},{"alias_kind":"pith_short_8","alias_value":"FTIMO5DT","created_at":"2026-07-05T09:08:32.330343+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.10401","citing_title":"NameBERT: Scaling Name-Based Nationality Classification with LLM-Augmented Open Academic Data","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FTIMO5DTWDURLFGUSBJH7BU4JW","json":"https://pith.science/pith/FTIMO5DTWDURLFGUSBJH7BU4JW.json","graph_json":"https://pith.science/api/pith-number/FTIMO5DTWDURLFGUSBJH7BU4JW/graph.json","events_json":"https://pith.science/api/pith-number/FTIMO5DTWDURLFGUSBJH7BU4JW/events.json","paper":"https://pith.science/paper/FTIMO5DT"},"agent_actions":{"view_html":"https://pith.science/pith/FTIMO5DTWDURLFGUSBJH7BU4JW","download_json":"https://pith.science/pith/FTIMO5DTWDURLFGUSBJH7BU4JW.json","view_paper":"https://pith.science/paper/FTIMO5DT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.11491&json=true","fetch_graph":"https://pith.science/api/pith-number/FTIMO5DTWDURLFGUSBJH7BU4JW/graph.json","fetch_events":"https://pith.science/api/pith-number/FTIMO5DTWDURLFGUSBJH7BU4JW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FTIMO5DTWDURLFGUSBJH7BU4JW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FTIMO5DTWDURLFGUSBJH7BU4JW/action/storage_attestation","attest_author":"https://pith.science/pith/FTIMO5DTWDURLFGUSBJH7BU4JW/action/author_attestation","sign_citation":"https://pith.science/pith/FTIMO5DTWDURLFGUSBJH7BU4JW/action/citation_signature","submit_replication":"https://pith.science/pith/FTIMO5DTWDURLFGUSBJH7BU4JW/action/replication_record"}},"created_at":"2026-07-05T09:08:32.330343+00:00","updated_at":"2026-07-05T09:08:32.330343+00:00"}