{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:CIBYMC5U4HBJCFO43N2T5XOJT6","short_pith_number":"pith:CIBYMC5U","schema_version":"1.0","canonical_sha256":"1203860bb4e1c29115dcdb753eddc99fbd6325295ea152c626a5a7fe0cbb7477","source":{"kind":"arxiv","id":"2408.01354","version":2},"attestation_state":"computed","paper":{"title":"MCGMark: An Encodable and Robust Online Watermark for Tracing LLM-Generated Malicious Code","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SE"],"primary_cat":"cs.CR","authors_text":"Jiachi Chen, Jianxing Yu, Jingwen Zhang, Kaiwen Ning, Qingyuan Zhong, Tao Zhang, Wei Li, Weizhe Zhang, Yanlin Wang, Yuming Feng, Zibin Zheng","submitted_at":"2024-08-02T16:04:52Z","abstract_excerpt":"With the advent of large language models (LLMs), numerous software service providers (SSPs) are dedicated to developing LLMs customized for code generation tasks, such as CodeLlama and Copilot. However, these LLMs can be leveraged by attackers to create malicious software, which may pose potential threats to the software ecosystem. For example, they can automate the creation of advanced phishing malware. To address this issue, we first conduct an empirical study and design a prompt dataset, MCGTest, which involves approximately 400 person-hours of work and consists of 406 malicious code genera"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.01354","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2024-08-02T16:04:52Z","cross_cats_sorted":["cs.SE"],"title_canon_sha256":"4dd25a022979aca6f811c45894e9ffaa5f8395f235f346f407f02efd89846ec0","abstract_canon_sha256":"e2646f198ddc5d05ef65889993cea73da93fa4594a9629e1a647e7cec9333084"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:51:24.380817Z","signature_b64":"fl0Jec5fSisgGFj4Ixen9+Vf09Vv3sxHAPkIdRgV71ehuPIfthyqtBIjEa9jXpdOTTz4ZJsEzCg8FdbtU418AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1203860bb4e1c29115dcdb753eddc99fbd6325295ea152c626a5a7fe0cbb7477","last_reissued_at":"2026-07-05T10:51:24.380307Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:51:24.380307Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MCGMark: An Encodable and Robust Online Watermark for Tracing LLM-Generated Malicious Code","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SE"],"primary_cat":"cs.CR","authors_text":"Jiachi Chen, Jianxing Yu, Jingwen Zhang, Kaiwen Ning, Qingyuan Zhong, Tao Zhang, Wei Li, Weizhe Zhang, Yanlin Wang, Yuming Feng, Zibin Zheng","submitted_at":"2024-08-02T16:04:52Z","abstract_excerpt":"With the advent of large language models (LLMs), numerous software service providers (SSPs) are dedicated to developing LLMs customized for code generation tasks, such as CodeLlama and Copilot. However, these LLMs can be leveraged by attackers to create malicious software, which may pose potential threats to the software ecosystem. For example, they can automate the creation of advanced phishing malware. To address this issue, we first conduct an empirical study and design a prompt dataset, MCGTest, which involves approximately 400 person-hours of work and consists of 406 malicious code genera"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.01354","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.01354/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.01354","created_at":"2026-07-05T10:51:24.380373+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.01354v2","created_at":"2026-07-05T10:51:24.380373+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.01354","created_at":"2026-07-05T10:51:24.380373+00:00"},{"alias_kind":"pith_short_12","alias_value":"CIBYMC5U4HBJ","created_at":"2026-07-05T10:51:24.380373+00:00"},{"alias_kind":"pith_short_16","alias_value":"CIBYMC5U4HBJCFO4","created_at":"2026-07-05T10:51:24.380373+00:00"},{"alias_kind":"pith_short_8","alias_value":"CIBYMC5U","created_at":"2026-07-05T10:51:24.380373+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20835","citing_title":"PromptMark: A Prompt-Guided Iterative-Feedback Framework for Source Code Watermarking","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20351","citing_title":"Refusal Evaluation in Coding LLMs and Code Agents: A Systematic Review of Thirteen Malicious-Code Prompt Corpora (2023-2025)","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16001","citing_title":"MATRIX: Multi-Layer Code Watermarking via Dual-Channel Constrained Parity-Check Encoding","ref_index":49,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CIBYMC5U4HBJCFO43N2T5XOJT6","json":"https://pith.science/pith/CIBYMC5U4HBJCFO43N2T5XOJT6.json","graph_json":"https://pith.science/api/pith-number/CIBYMC5U4HBJCFO43N2T5XOJT6/graph.json","events_json":"https://pith.science/api/pith-number/CIBYMC5U4HBJCFO43N2T5XOJT6/events.json","paper":"https://pith.science/paper/CIBYMC5U"},"agent_actions":{"view_html":"https://pith.science/pith/CIBYMC5U4HBJCFO43N2T5XOJT6","download_json":"https://pith.science/pith/CIBYMC5U4HBJCFO43N2T5XOJT6.json","view_paper":"https://pith.science/paper/CIBYMC5U","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.01354&json=true","fetch_graph":"https://pith.science/api/pith-number/CIBYMC5U4HBJCFO43N2T5XOJT6/graph.json","fetch_events":"https://pith.science/api/pith-number/CIBYMC5U4HBJCFO43N2T5XOJT6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CIBYMC5U4HBJCFO43N2T5XOJT6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CIBYMC5U4HBJCFO43N2T5XOJT6/action/storage_attestation","attest_author":"https://pith.science/pith/CIBYMC5U4HBJCFO43N2T5XOJT6/action/author_attestation","sign_citation":"https://pith.science/pith/CIBYMC5U4HBJCFO43N2T5XOJT6/action/citation_signature","submit_replication":"https://pith.science/pith/CIBYMC5U4HBJCFO43N2T5XOJT6/action/replication_record"}},"created_at":"2026-07-05T10:51:24.380373+00:00","updated_at":"2026-07-05T10:51:24.380373+00:00"}