{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2018:RLAUC45M2KFW5IBB2DP6OFGKXG","short_pith_number":"pith:RLAUC45M","schema_version":"1.0","canonical_sha256":"8ac14173acd28b6ea021d0dfe714cab9887539e1ec21b8b677305bd3048ff385","source":{"kind":"arxiv","id":"1812.10424","version":4},"attestation_state":"computed","paper":{"title":"Measuring Societal Biases from Text Corpora with Smoothed First-Order Co-occurrence","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.CL","authors_text":"Allan Hanbury, James Henderson, Navid Rekabsaz, Robert West","submitted_at":"2018-12-13T21:00:05Z","abstract_excerpt":"Text corpora are widely used resources for measuring societal biases and stereotypes. The common approach to measuring such biases using a corpus is by calculating the similarities between the embedding vector of a word (like nurse) and the vectors of the representative words of the concepts of interest (such as genders). In this study, we show that, depending on what one aims to quantify as bias, this commonly-used approach can introduce non-relevant concepts into bias measurement. We propose an alternative approach to bias measurement utilizing the smoothed first-order co-occurrence relation"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1812.10424","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2018-12-13T21:00:05Z","cross_cats_sorted":["cs.LG","stat.ML"],"title_canon_sha256":"4316c9619d106961a862047c2afe26b56ddc256fd6e2649f4c64f553dfb415b2","abstract_canon_sha256":"9ae10687e1eba279455e861bcbaa83fb5c426ded7c768d9a419b16054a54dc9c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:35:09.914908Z","signature_b64":"Z/jUAN9wRlrSQavMyLrwVykZa2ZYj6/YZ2PeV7tI5a5KR5wWFAkrHFRcTWXRtxM+C6wVEKouQK8wOt5HyusWCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8ac14173acd28b6ea021d0dfe714cab9887539e1ec21b8b677305bd3048ff385","last_reissued_at":"2026-07-05T02:35:09.914439Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:35:09.914439Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Measuring Societal Biases from Text Corpora with Smoothed First-Order Co-occurrence","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.CL","authors_text":"Allan Hanbury, James Henderson, Navid Rekabsaz, Robert West","submitted_at":"2018-12-13T21:00:05Z","abstract_excerpt":"Text corpora are widely used resources for measuring societal biases and stereotypes. The common approach to measuring such biases using a corpus is by calculating the similarities between the embedding vector of a word (like nurse) and the vectors of the representative words of the concepts of interest (such as genders). In this study, we show that, depending on what one aims to quantify as bias, this commonly-used approach can introduce non-relevant concepts into bias measurement. We propose an alternative approach to bias measurement utilizing the smoothed first-order co-occurrence relation"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1812.10424","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1812.10424/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1812.10424","created_at":"2026-07-05T02:35:09.914497+00:00"},{"alias_kind":"arxiv_version","alias_value":"1812.10424v4","created_at":"2026-07-05T02:35:09.914497+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1812.10424","created_at":"2026-07-05T02:35:09.914497+00:00"},{"alias_kind":"pith_short_12","alias_value":"RLAUC45M2KFW","created_at":"2026-07-05T02:35:09.914497+00:00"},{"alias_kind":"pith_short_16","alias_value":"RLAUC45M2KFW5IBB","created_at":"2026-07-05T02:35:09.914497+00:00"},{"alias_kind":"pith_short_8","alias_value":"RLAUC45M","created_at":"2026-07-05T02:35:09.914497+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.26062","citing_title":"Identifying Implicit Bias in LLM-based Chat AI Toward People with Intellectual Disabilities","ref_index":17,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RLAUC45M2KFW5IBB2DP6OFGKXG","json":"https://pith.science/pith/RLAUC45M2KFW5IBB2DP6OFGKXG.json","graph_json":"https://pith.science/api/pith-number/RLAUC45M2KFW5IBB2DP6OFGKXG/graph.json","events_json":"https://pith.science/api/pith-number/RLAUC45M2KFW5IBB2DP6OFGKXG/events.json","paper":"https://pith.science/paper/RLAUC45M"},"agent_actions":{"view_html":"https://pith.science/pith/RLAUC45M2KFW5IBB2DP6OFGKXG","download_json":"https://pith.science/pith/RLAUC45M2KFW5IBB2DP6OFGKXG.json","view_paper":"https://pith.science/paper/RLAUC45M","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1812.10424&json=true","fetch_graph":"https://pith.science/api/pith-number/RLAUC45M2KFW5IBB2DP6OFGKXG/graph.json","fetch_events":"https://pith.science/api/pith-number/RLAUC45M2KFW5IBB2DP6OFGKXG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RLAUC45M2KFW5IBB2DP6OFGKXG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RLAUC45M2KFW5IBB2DP6OFGKXG/action/storage_attestation","attest_author":"https://pith.science/pith/RLAUC45M2KFW5IBB2DP6OFGKXG/action/author_attestation","sign_citation":"https://pith.science/pith/RLAUC45M2KFW5IBB2DP6OFGKXG/action/citation_signature","submit_replication":"https://pith.science/pith/RLAUC45M2KFW5IBB2DP6OFGKXG/action/replication_record"}},"created_at":"2026-07-05T02:35:09.914497+00:00","updated_at":"2026-07-05T02:35:09.914497+00:00"}