{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:DBY2YMV6W7ZLOCTSM7IEYLPNJZ","short_pith_number":"pith:DBY2YMV6","schema_version":"1.0","canonical_sha256":"1871ac32beb7f2b70a7267d04c2ded4e4cfe0dfcee006b3405322080559223a2","source":{"kind":"arxiv","id":"2306.13141","version":2},"attestation_state":"computed","paper":{"title":"On Hate Scaling Laws For Data-Swamps","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CY","authors_text":"Abeba Birhane, Sang Han, Vinay Prabhu, Vishnu Naresh Boddeti","submitted_at":"2023-06-22T18:00:17Z","abstract_excerpt":"`Scale the model, scale the data, scale the GPU-farms' is the reigning sentiment in the world of generative AI today. While model scaling has been extensively studied, data scaling and its downstream impacts remain under explored. This is especially of critical importance in the context of visio-linguistic datasets whose main source is the World Wide Web, condensed and packaged as the CommonCrawl dump. This large scale data-dump, which is known to have numerous drawbacks, is repeatedly mined and serves as the data-motherlode for large generative models. In this paper, we: 1) investigate the ef"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.13141","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CY","submitted_at":"2023-06-22T18:00:17Z","cross_cats_sorted":[],"title_canon_sha256":"408e7d37239d2151f892249aa06a6e5fad2cb3399265a2402bdb3cf16f7f76d1","abstract_canon_sha256":"31a56bc56d59916b515b0d3813c9eafe6a6bb0e1b646b47e709db358b45dc5f5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:26:03.279438Z","signature_b64":"vABFwE/KSzy02T5nEcekmmHd3qIecZzNNHmZQYpauc2cwHRBbHxfSsB6hSIKu3El1tJqgJi9KFRi8c0NIz8TDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1871ac32beb7f2b70a7267d04c2ded4e4cfe0dfcee006b3405322080559223a2","last_reissued_at":"2026-07-05T06:26:03.278968Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:26:03.278968Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On Hate Scaling Laws For Data-Swamps","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CY","authors_text":"Abeba Birhane, Sang Han, Vinay Prabhu, Vishnu Naresh Boddeti","submitted_at":"2023-06-22T18:00:17Z","abstract_excerpt":"`Scale the model, scale the data, scale the GPU-farms' is the reigning sentiment in the world of generative AI today. While model scaling has been extensively studied, data scaling and its downstream impacts remain under explored. This is especially of critical importance in the context of visio-linguistic datasets whose main source is the World Wide Web, condensed and packaged as the CommonCrawl dump. This large scale data-dump, which is known to have numerous drawbacks, is repeatedly mined and serves as the data-motherlode for large generative models. In this paper, we: 1) investigate the ef"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.13141","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.13141/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.13141","created_at":"2026-07-05T06:26:03.279027+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.13141v2","created_at":"2026-07-05T06:26:03.279027+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.13141","created_at":"2026-07-05T06:26:03.279027+00:00"},{"alias_kind":"pith_short_12","alias_value":"DBY2YMV6W7ZL","created_at":"2026-07-05T06:26:03.279027+00:00"},{"alias_kind":"pith_short_16","alias_value":"DBY2YMV6W7ZLOCTS","created_at":"2026-07-05T06:26:03.279027+00:00"},{"alias_kind":"pith_short_8","alias_value":"DBY2YMV6","created_at":"2026-07-05T06:26:03.279027+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.30800","citing_title":"Computer-Aided Tagging on Wikimedia Commons: Designing for Human-AI Collaboration in Open Knowledge Work","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13305","citing_title":"Bias at the End of the Score","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DBY2YMV6W7ZLOCTSM7IEYLPNJZ","json":"https://pith.science/pith/DBY2YMV6W7ZLOCTSM7IEYLPNJZ.json","graph_json":"https://pith.science/api/pith-number/DBY2YMV6W7ZLOCTSM7IEYLPNJZ/graph.json","events_json":"https://pith.science/api/pith-number/DBY2YMV6W7ZLOCTSM7IEYLPNJZ/events.json","paper":"https://pith.science/paper/DBY2YMV6"},"agent_actions":{"view_html":"https://pith.science/pith/DBY2YMV6W7ZLOCTSM7IEYLPNJZ","download_json":"https://pith.science/pith/DBY2YMV6W7ZLOCTSM7IEYLPNJZ.json","view_paper":"https://pith.science/paper/DBY2YMV6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.13141&json=true","fetch_graph":"https://pith.science/api/pith-number/DBY2YMV6W7ZLOCTSM7IEYLPNJZ/graph.json","fetch_events":"https://pith.science/api/pith-number/DBY2YMV6W7ZLOCTSM7IEYLPNJZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DBY2YMV6W7ZLOCTSM7IEYLPNJZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DBY2YMV6W7ZLOCTSM7IEYLPNJZ/action/storage_attestation","attest_author":"https://pith.science/pith/DBY2YMV6W7ZLOCTSM7IEYLPNJZ/action/author_attestation","sign_citation":"https://pith.science/pith/DBY2YMV6W7ZLOCTSM7IEYLPNJZ/action/citation_signature","submit_replication":"https://pith.science/pith/DBY2YMV6W7ZLOCTSM7IEYLPNJZ/action/replication_record"}},"created_at":"2026-07-05T06:26:03.279027+00:00","updated_at":"2026-07-05T06:26:03.279027+00:00"}