{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:C2UPA4FNH3UEIVZBURQTSZHRWO","short_pith_number":"pith:C2UPA4FN","schema_version":"1.0","canonical_sha256":"16a8f070ad3ee8445721a4613964f1b3bff1764a28e9568fdea582a940f4f4ff","source":{"kind":"arxiv","id":"2302.07388","version":1},"attestation_state":"computed","paper":{"title":"Adding Instructions during Pretraining: Effective Way of Controlling Toxicity in Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bryan Catanzaro, Mohammad Shoeybi, Mostofa Patwary, Shrimai Prabhumoye","submitted_at":"2023-02-14T23:00:42Z","abstract_excerpt":"Pretrained large language models have become indispensable for solving various natural language processing (NLP) tasks. However, safely deploying them in real world applications is challenging because they generate toxic content. To address this challenge, we propose two novel pretraining data augmentation strategies that significantly reduce model toxicity without compromising its utility. Our two strategies are: (1) MEDA: adds raw toxicity score as meta-data to the pretraining samples, and (2) INST: adds instructions to those samples indicating their toxicity. Our results indicate that our b"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2302.07388","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-02-14T23:00:42Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"3fa68d5b30ace192165730b31f05864004e8a862c8e73cf736a4808e8058675c","abstract_canon_sha256":"c60d67fa7d623dbffdd74c8da20d2b337edeba3c476f03ae52ce6f6620f5008a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:41:53.619275Z","signature_b64":"gs4dQQCxPsv5A5EIQbsc682Q7LWjvCLA47q/EBVt8YG1y+iHbCO8oKJ40eMEMsRNc6mffaylIFLhGfAX+IAWCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"16a8f070ad3ee8445721a4613964f1b3bff1764a28e9568fdea582a940f4f4ff","last_reissued_at":"2026-07-05T05:41:53.618802Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:41:53.618802Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Adding Instructions during Pretraining: Effective Way of Controlling Toxicity in Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bryan Catanzaro, Mohammad Shoeybi, Mostofa Patwary, Shrimai Prabhumoye","submitted_at":"2023-02-14T23:00:42Z","abstract_excerpt":"Pretrained large language models have become indispensable for solving various natural language processing (NLP) tasks. However, safely deploying them in real world applications is challenging because they generate toxic content. To address this challenge, we propose two novel pretraining data augmentation strategies that significantly reduce model toxicity without compromising its utility. Our two strategies are: (1) MEDA: adds raw toxicity score as meta-data to the pretraining samples, and (2) INST: adds instructions to those samples indicating their toxicity. Our results indicate that our b"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.07388","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.07388/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2302.07388","created_at":"2026-07-05T05:41:53.618862+00:00"},{"alias_kind":"arxiv_version","alias_value":"2302.07388v1","created_at":"2026-07-05T05:41:53.618862+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.07388","created_at":"2026-07-05T05:41:53.618862+00:00"},{"alias_kind":"pith_short_12","alias_value":"C2UPA4FNH3UE","created_at":"2026-07-05T05:41:53.618862+00:00"},{"alias_kind":"pith_short_16","alias_value":"C2UPA4FNH3UEIVZB","created_at":"2026-07-05T05:41:53.618862+00:00"},{"alias_kind":"pith_short_8","alias_value":"C2UPA4FN","created_at":"2026-07-05T05:41:53.618862+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2305.16264","citing_title":"Scaling Data-Constrained Language Models","ref_index":94,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/C2UPA4FNH3UEIVZBURQTSZHRWO","json":"https://pith.science/pith/C2UPA4FNH3UEIVZBURQTSZHRWO.json","graph_json":"https://pith.science/api/pith-number/C2UPA4FNH3UEIVZBURQTSZHRWO/graph.json","events_json":"https://pith.science/api/pith-number/C2UPA4FNH3UEIVZBURQTSZHRWO/events.json","paper":"https://pith.science/paper/C2UPA4FN"},"agent_actions":{"view_html":"https://pith.science/pith/C2UPA4FNH3UEIVZBURQTSZHRWO","download_json":"https://pith.science/pith/C2UPA4FNH3UEIVZBURQTSZHRWO.json","view_paper":"https://pith.science/paper/C2UPA4FN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2302.07388&json=true","fetch_graph":"https://pith.science/api/pith-number/C2UPA4FNH3UEIVZBURQTSZHRWO/graph.json","fetch_events":"https://pith.science/api/pith-number/C2UPA4FNH3UEIVZBURQTSZHRWO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/C2UPA4FNH3UEIVZBURQTSZHRWO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/C2UPA4FNH3UEIVZBURQTSZHRWO/action/storage_attestation","attest_author":"https://pith.science/pith/C2UPA4FNH3UEIVZBURQTSZHRWO/action/author_attestation","sign_citation":"https://pith.science/pith/C2UPA4FNH3UEIVZBURQTSZHRWO/action/citation_signature","submit_replication":"https://pith.science/pith/C2UPA4FNH3UEIVZBURQTSZHRWO/action/replication_record"}},"created_at":"2026-07-05T05:41:53.618862+00:00","updated_at":"2026-07-05T05:41:53.618862+00:00"}