{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:PSMCY4DQHPHP3QZQPWDLKTOPIC","short_pith_number":"pith:PSMCY4DQ","schema_version":"1.0","canonical_sha256":"7c982c70703bcefdc3307d86b54dcf40924e284cb4740cb769eedfbf171aaec1","source":{"kind":"arxiv","id":"2310.14429","version":1},"attestation_state":"computed","paper":{"title":"Text generation for dataset augmentation in security classification tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CR","authors_text":"Alexander P. Welsh, Matthew Edwards","submitted_at":"2023-10-22T22:25:14Z","abstract_excerpt":"Security classifiers, designed to detect malicious content in computer systems and communications, can underperform when provided with insufficient training data. In the security domain, it is often easy to find samples of the negative (benign) class, and challenging to find enough samples of the positive (malicious) class to train an effective classifier. This study evaluates the application of natural language text generators to fill this data gap in multiple security-related text classification tasks. We describe a variety of previously-unexamined language-model fine-tuning approaches for t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.14429","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2023-10-22T22:25:14Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"0188b9dc2c1f6cf68694811dfe1c078858ffa30117fddc269c8a42f4bf7ea2be","abstract_canon_sha256":"e67683c8bed36a101c264199ea0d27145e03aab9210b75d71203fd9b8aef1047"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:03:44.243346Z","signature_b64":"DQt1anjrYfmTbz64edu89/DIPX7DITaSL09JI5bzvfXNk5XgJ70IKJC63x+poRYIVGfyGhg/7V2ct9BbNPglAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7c982c70703bcefdc3307d86b54dcf40924e284cb4740cb769eedfbf171aaec1","last_reissued_at":"2026-07-05T07:03:44.242867Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:03:44.242867Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Text generation for dataset augmentation in security classification tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CR","authors_text":"Alexander P. Welsh, Matthew Edwards","submitted_at":"2023-10-22T22:25:14Z","abstract_excerpt":"Security classifiers, designed to detect malicious content in computer systems and communications, can underperform when provided with insufficient training data. In the security domain, it is often easy to find samples of the negative (benign) class, and challenging to find enough samples of the positive (malicious) class to train an effective classifier. This study evaluates the application of natural language text generators to fill this data gap in multiple security-related text classification tasks. We describe a variety of previously-unexamined language-model fine-tuning approaches for t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.14429","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.14429/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.14429","created_at":"2026-07-05T07:03:44.242923+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.14429v1","created_at":"2026-07-05T07:03:44.242923+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.14429","created_at":"2026-07-05T07:03:44.242923+00:00"},{"alias_kind":"pith_short_12","alias_value":"PSMCY4DQHPHP","created_at":"2026-07-05T07:03:44.242923+00:00"},{"alias_kind":"pith_short_16","alias_value":"PSMCY4DQHPHP3QZQ","created_at":"2026-07-05T07:03:44.242923+00:00"},{"alias_kind":"pith_short_8","alias_value":"PSMCY4DQ","created_at":"2026-07-05T07:03:44.242923+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PSMCY4DQHPHP3QZQPWDLKTOPIC","json":"https://pith.science/pith/PSMCY4DQHPHP3QZQPWDLKTOPIC.json","graph_json":"https://pith.science/api/pith-number/PSMCY4DQHPHP3QZQPWDLKTOPIC/graph.json","events_json":"https://pith.science/api/pith-number/PSMCY4DQHPHP3QZQPWDLKTOPIC/events.json","paper":"https://pith.science/paper/PSMCY4DQ"},"agent_actions":{"view_html":"https://pith.science/pith/PSMCY4DQHPHP3QZQPWDLKTOPIC","download_json":"https://pith.science/pith/PSMCY4DQHPHP3QZQPWDLKTOPIC.json","view_paper":"https://pith.science/paper/PSMCY4DQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.14429&json=true","fetch_graph":"https://pith.science/api/pith-number/PSMCY4DQHPHP3QZQPWDLKTOPIC/graph.json","fetch_events":"https://pith.science/api/pith-number/PSMCY4DQHPHP3QZQPWDLKTOPIC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PSMCY4DQHPHP3QZQPWDLKTOPIC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PSMCY4DQHPHP3QZQPWDLKTOPIC/action/storage_attestation","attest_author":"https://pith.science/pith/PSMCY4DQHPHP3QZQPWDLKTOPIC/action/author_attestation","sign_citation":"https://pith.science/pith/PSMCY4DQHPHP3QZQPWDLKTOPIC/action/citation_signature","submit_replication":"https://pith.science/pith/PSMCY4DQHPHP3QZQPWDLKTOPIC/action/replication_record"}},"created_at":"2026-07-05T07:03:44.242923+00:00","updated_at":"2026-07-05T07:03:44.242923+00:00"}