{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:EIDBUXB3LJX43WXW3BBCMN5VM5","short_pith_number":"pith:EIDBUXB3","schema_version":"1.0","canonical_sha256":"22061a5c3b5a6fcddaf6d8422637b567799c4d4b441b0728bfd8f8c2fbb63fe1","source":{"kind":"arxiv","id":"2004.01670","version":3},"attestation_state":"computed","paper":{"title":"Directions in Abusive Language Training Data: Garbage In, Garbage Out","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bertie Vidgen, Leon Derczynski","submitted_at":"2020-04-03T16:51:33Z","abstract_excerpt":"Data-driven analysis and detection of abusive online content covers many different tasks, phenomena, contexts, and methodologies. This paper systematically reviews abusive language dataset creation and content in conjunction with an open website for cataloguing abusive language data. This collection of knowledge leads to a synthesis providing evidence-based recommendations for practitioners working with this complex and highly diverse data."},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2004.01670","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-04-03T16:51:33Z","cross_cats_sorted":[],"title_canon_sha256":"aa3aa59e902979d88f4c9f26e79ad427705a3d80997db9435d653d8fdddc37d4","abstract_canon_sha256":"eafebd88cbdf1781120230cf8ccc836fffab65cdfafcc2b64a9982e5417e7ad4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:53:45.755197Z","signature_b64":"ay9FWRt78G92SNleVrTaYjXmCnliLHMa+HohYkTWnBFldU6hBcsXC+ic7s71RNeOVWU5VWUywNX3smouHE36Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"22061a5c3b5a6fcddaf6d8422637b567799c4d4b441b0728bfd8f8c2fbb63fe1","last_reissued_at":"2026-07-05T05:53:45.754850Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:53:45.754850Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Directions in Abusive Language Training Data: Garbage In, Garbage Out","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bertie Vidgen, Leon Derczynski","submitted_at":"2020-04-03T16:51:33Z","abstract_excerpt":"Data-driven analysis and detection of abusive online content covers many different tasks, phenomena, contexts, and methodologies. This paper systematically reviews abusive language dataset creation and content in conjunction with an open website for cataloguing abusive language data. This collection of knowledge leads to a synthesis providing evidence-based recommendations for practitioners working with this complex and highly diverse data."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2004.01670","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2004.01670/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2004.01670","created_at":"2026-07-05T05:53:45.754907+00:00"},{"alias_kind":"arxiv_version","alias_value":"2004.01670v3","created_at":"2026-07-05T05:53:45.754907+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2004.01670","created_at":"2026-07-05T05:53:45.754907+00:00"},{"alias_kind":"pith_short_12","alias_value":"EIDBUXB3LJX4","created_at":"2026-07-05T05:53:45.754907+00:00"},{"alias_kind":"pith_short_16","alias_value":"EIDBUXB3LJX43WXW","created_at":"2026-07-05T05:53:45.754907+00:00"},{"alias_kind":"pith_short_8","alias_value":"EIDBUXB3","created_at":"2026-07-05T05:53:45.754907+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.00713","citing_title":"CODEOFCONDUCT at Multilingual Counterspeech Generation: A Context-Aware Model for Robust Counterspeech Generation in Low-Resource Languages","ref_index":14,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EIDBUXB3LJX43WXW3BBCMN5VM5","json":"https://pith.science/pith/EIDBUXB3LJX43WXW3BBCMN5VM5.json","graph_json":"https://pith.science/api/pith-number/EIDBUXB3LJX43WXW3BBCMN5VM5/graph.json","events_json":"https://pith.science/api/pith-number/EIDBUXB3LJX43WXW3BBCMN5VM5/events.json","paper":"https://pith.science/paper/EIDBUXB3"},"agent_actions":{"view_html":"https://pith.science/pith/EIDBUXB3LJX43WXW3BBCMN5VM5","download_json":"https://pith.science/pith/EIDBUXB3LJX43WXW3BBCMN5VM5.json","view_paper":"https://pith.science/paper/EIDBUXB3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2004.01670&json=true","fetch_graph":"https://pith.science/api/pith-number/EIDBUXB3LJX43WXW3BBCMN5VM5/graph.json","fetch_events":"https://pith.science/api/pith-number/EIDBUXB3LJX43WXW3BBCMN5VM5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EIDBUXB3LJX43WXW3BBCMN5VM5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EIDBUXB3LJX43WXW3BBCMN5VM5/action/storage_attestation","attest_author":"https://pith.science/pith/EIDBUXB3LJX43WXW3BBCMN5VM5/action/author_attestation","sign_citation":"https://pith.science/pith/EIDBUXB3LJX43WXW3BBCMN5VM5/action/citation_signature","submit_replication":"https://pith.science/pith/EIDBUXB3LJX43WXW3BBCMN5VM5/action/replication_record"}},"created_at":"2026-07-05T05:53:45.754907+00:00","updated_at":"2026-07-05T05:53:45.754907+00:00"}