{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:F7CA24JTADDEWYGZEDCQW76DSW","short_pith_number":"pith:F7CA24JT","schema_version":"1.0","canonical_sha256":"2fc40d713300c64b60d920c50b7fc395812e592a819f6730f736693090832cd2","source":{"kind":"arxiv","id":"2506.13513","version":1},"attestation_state":"computed","paper":{"title":"K/DA: Automated Data Generation Pipeline for Detoxifying Implicitly Offensive Language in Korean","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Byung-Jun Lee, Hyemin Jeong, Jae Hyeon Cho, Jiyoung Kim, Minkyeong Jeon, Yerang Kim","submitted_at":"2025-06-16T14:08:23Z","abstract_excerpt":"Language detoxification involves removing toxicity from offensive language. While a neutral-toxic paired dataset provides a straightforward approach for training detoxification models, creating such datasets presents several challenges: i) the need for human annotation to build paired data, and ii) the rapid evolution of offensive terms, rendering static datasets quickly outdated. To tackle these challenges, we introduce an automated paired data generation pipeline, called K/DA. This pipeline is designed to generate offensive language with implicit offensiveness and trend-aligned slang, making"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.13513","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-16T14:08:23Z","cross_cats_sorted":[],"title_canon_sha256":"d1fee88b6f4c280dcc483084e11fb591f0ae84ca891f7b22e32d171d00662270","abstract_canon_sha256":"cf7a052c9d14cccdd2f56028dd6e384f9831c6aa2b34de4bebef81050c77db74"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:22:21.435807Z","signature_b64":"gAf7ykUJJmsSCAp8knryax68bBTL5HsHic9O8OUPoFMcQhIxdvjZFA0Kpb7ORl319gfhSuLJwoUA4G81btFkDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2fc40d713300c64b60d920c50b7fc395812e592a819f6730f736693090832cd2","last_reissued_at":"2026-07-05T11:22:21.435290Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:22:21.435290Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"K/DA: Automated Data Generation Pipeline for Detoxifying Implicitly Offensive Language in Korean","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Byung-Jun Lee, Hyemin Jeong, Jae Hyeon Cho, Jiyoung Kim, Minkyeong Jeon, Yerang Kim","submitted_at":"2025-06-16T14:08:23Z","abstract_excerpt":"Language detoxification involves removing toxicity from offensive language. While a neutral-toxic paired dataset provides a straightforward approach for training detoxification models, creating such datasets presents several challenges: i) the need for human annotation to build paired data, and ii) the rapid evolution of offensive terms, rendering static datasets quickly outdated. To tackle these challenges, we introduce an automated paired data generation pipeline, called K/DA. This pipeline is designed to generate offensive language with implicit offensiveness and trend-aligned slang, making"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.13513","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.13513/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.13513","created_at":"2026-07-05T11:22:21.435360+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.13513v1","created_at":"2026-07-05T11:22:21.435360+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.13513","created_at":"2026-07-05T11:22:21.435360+00:00"},{"alias_kind":"pith_short_12","alias_value":"F7CA24JTADDE","created_at":"2026-07-05T11:22:21.435360+00:00"},{"alias_kind":"pith_short_16","alias_value":"F7CA24JTADDEWYGZ","created_at":"2026-07-05T11:22:21.435360+00:00"},{"alias_kind":"pith_short_8","alias_value":"F7CA24JT","created_at":"2026-07-05T11:22:21.435360+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.17020","citing_title":"Beyond Static Benchmarks: Synthesizing Harmful Content via Persona-based Simulation for Robust Evaluation","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F7CA24JTADDEWYGZEDCQW76DSW","json":"https://pith.science/pith/F7CA24JTADDEWYGZEDCQW76DSW.json","graph_json":"https://pith.science/api/pith-number/F7CA24JTADDEWYGZEDCQW76DSW/graph.json","events_json":"https://pith.science/api/pith-number/F7CA24JTADDEWYGZEDCQW76DSW/events.json","paper":"https://pith.science/paper/F7CA24JT"},"agent_actions":{"view_html":"https://pith.science/pith/F7CA24JTADDEWYGZEDCQW76DSW","download_json":"https://pith.science/pith/F7CA24JTADDEWYGZEDCQW76DSW.json","view_paper":"https://pith.science/paper/F7CA24JT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.13513&json=true","fetch_graph":"https://pith.science/api/pith-number/F7CA24JTADDEWYGZEDCQW76DSW/graph.json","fetch_events":"https://pith.science/api/pith-number/F7CA24JTADDEWYGZEDCQW76DSW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F7CA24JTADDEWYGZEDCQW76DSW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F7CA24JTADDEWYGZEDCQW76DSW/action/storage_attestation","attest_author":"https://pith.science/pith/F7CA24JTADDEWYGZEDCQW76DSW/action/author_attestation","sign_citation":"https://pith.science/pith/F7CA24JTADDEWYGZEDCQW76DSW/action/citation_signature","submit_replication":"https://pith.science/pith/F7CA24JTADDEWYGZEDCQW76DSW/action/replication_record"}},"created_at":"2026-07-05T11:22:21.435360+00:00","updated_at":"2026-07-05T11:22:21.435360+00:00"}