{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:G3GYNY453DU6MDFE6YR3MF3ZDU","short_pith_number":"pith:G3GYNY45","schema_version":"1.0","canonical_sha256":"36cd86e39dd8e9e60ca4f623b617791d06e25986cd5f577ffed8184f275303dd","source":{"kind":"arxiv","id":"2112.03529","version":1},"attestation_state":"computed","paper":{"title":"Ground-Truth, Whose Truth? -- Examining the Challenges with Annotating Toxic Text Datasets","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dennis Wei, Ioana Baldini, Karthikeyan Natesan Ramamurthy, Kofi Arhin, Moninder Singh","submitted_at":"2021-12-07T06:58:22Z","abstract_excerpt":"The use of machine learning (ML)-based language models (LMs) to monitor content online is on the rise. For toxic text identification, task-specific fine-tuning of these models are performed using datasets labeled by annotators who provide ground-truth labels in an effort to distinguish between offensive and normal content. These projects have led to the development, improvement, and expansion of large datasets over time, and have contributed immensely to research on natural language. Despite the achievements, existing evidence suggests that ML models built on these datasets do not always resul"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2112.03529","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-12-07T06:58:22Z","cross_cats_sorted":[],"title_canon_sha256":"e4c6a06df4c08364269eab7c8a6d0c1f30c97da503c3bceaf07f8cc63f9c23be","abstract_canon_sha256":"8386ee0fa1ed3d79222d8e842528edc023442695d713942b3d6a95dfc8d33135"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:38:46.555327Z","signature_b64":"n/9LDxV8a1bL/dR2rCklLSpb9T19WG2+IcE08BpbbhLyJKJei63Ki5FywCS9xwCRqo4vp1ZvJiPkfCJh4ij2Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"36cd86e39dd8e9e60ca4f623b617791d06e25986cd5f577ffed8184f275303dd","last_reissued_at":"2026-07-05T03:38:46.554888Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:38:46.554888Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Ground-Truth, Whose Truth? -- Examining the Challenges with Annotating Toxic Text Datasets","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dennis Wei, Ioana Baldini, Karthikeyan Natesan Ramamurthy, Kofi Arhin, Moninder Singh","submitted_at":"2021-12-07T06:58:22Z","abstract_excerpt":"The use of machine learning (ML)-based language models (LMs) to monitor content online is on the rise. For toxic text identification, task-specific fine-tuning of these models are performed using datasets labeled by annotators who provide ground-truth labels in an effort to distinguish between offensive and normal content. These projects have led to the development, improvement, and expansion of large datasets over time, and have contributed immensely to research on natural language. Despite the achievements, existing evidence suggests that ML models built on these datasets do not always resul"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2112.03529","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2112.03529/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2112.03529","created_at":"2026-07-05T03:38:46.554943+00:00"},{"alias_kind":"arxiv_version","alias_value":"2112.03529v1","created_at":"2026-07-05T03:38:46.554943+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2112.03529","created_at":"2026-07-05T03:38:46.554943+00:00"},{"alias_kind":"pith_short_12","alias_value":"G3GYNY453DU6","created_at":"2026-07-05T03:38:46.554943+00:00"},{"alias_kind":"pith_short_16","alias_value":"G3GYNY453DU6MDFE","created_at":"2026-07-05T03:38:46.554943+00:00"},{"alias_kind":"pith_short_8","alias_value":"G3GYNY45","created_at":"2026-07-05T03:38:46.554943+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.07297","citing_title":"Data Annotation as Measurement","ref_index":2026,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/G3GYNY453DU6MDFE6YR3MF3ZDU","json":"https://pith.science/pith/G3GYNY453DU6MDFE6YR3MF3ZDU.json","graph_json":"https://pith.science/api/pith-number/G3GYNY453DU6MDFE6YR3MF3ZDU/graph.json","events_json":"https://pith.science/api/pith-number/G3GYNY453DU6MDFE6YR3MF3ZDU/events.json","paper":"https://pith.science/paper/G3GYNY45"},"agent_actions":{"view_html":"https://pith.science/pith/G3GYNY453DU6MDFE6YR3MF3ZDU","download_json":"https://pith.science/pith/G3GYNY453DU6MDFE6YR3MF3ZDU.json","view_paper":"https://pith.science/paper/G3GYNY45","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2112.03529&json=true","fetch_graph":"https://pith.science/api/pith-number/G3GYNY453DU6MDFE6YR3MF3ZDU/graph.json","fetch_events":"https://pith.science/api/pith-number/G3GYNY453DU6MDFE6YR3MF3ZDU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/G3GYNY453DU6MDFE6YR3MF3ZDU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/G3GYNY453DU6MDFE6YR3MF3ZDU/action/storage_attestation","attest_author":"https://pith.science/pith/G3GYNY453DU6MDFE6YR3MF3ZDU/action/author_attestation","sign_citation":"https://pith.science/pith/G3GYNY453DU6MDFE6YR3MF3ZDU/action/citation_signature","submit_replication":"https://pith.science/pith/G3GYNY453DU6MDFE6YR3MF3ZDU/action/replication_record"}},"created_at":"2026-07-05T03:38:46.554943+00:00","updated_at":"2026-07-05T03:38:46.554943+00:00"}