{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:54MPQKZOWCV3WPTYN3ZBVL7PDO","short_pith_number":"pith:54MPQKZO","schema_version":"1.0","canonical_sha256":"ef18f82b2eb0abbb3e786ef21aafef1b80daadfa7ef6734ac335d5adc13eb719","source":{"kind":"arxiv","id":"2404.14296","version":2},"attestation_state":"computed","paper":{"title":"Does Your Neural Code Completion Model Use My Code? A Membership Inference Approach","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SE","authors_text":"Guanghua Wan, Hai Jin, Hongyu Zhang, Lichao Sun, Pan Zhou, Shijie Zhang, Yao Wan","submitted_at":"2024-04-22T15:54:53Z","abstract_excerpt":"Recent years have witnessed significant progress in developing deep learning-based models for automated code completion. Although using source code in GitHub has been a common practice for training deep-learning-based models for code completion, it may induce some legal and ethical issues such as copyright infringement. In this paper, we investigate the legal and ethical issues of current neural code completion models by answering the following question: Is my code used to train your neural code completion model? To this end, we tailor a membership inference approach (termed CodeMI) that was o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.14296","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2024-04-22T15:54:53Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"fc70fe07f98fc56bc8833c3c8fc6c0eb9834ce7d582fd40a858852ace69db080","abstract_canon_sha256":"b818bb70aee115798c00f665608293e8eaff78deaa6748cda2838470f701f72b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:04:09.412118Z","signature_b64":"bNkLLDUd0xeiFmFCBmIkYUI7OQVMyUJWPumXv2itYW19SwDEWkb8zKz0vvCw6+ycZ31nZtvvxf4a/V0YxIARBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ef18f82b2eb0abbb3e786ef21aafef1b80daadfa7ef6734ac335d5adc13eb719","last_reissued_at":"2026-07-05T09:04:09.411653Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:04:09.411653Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Does Your Neural Code Completion Model Use My Code? A Membership Inference Approach","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SE","authors_text":"Guanghua Wan, Hai Jin, Hongyu Zhang, Lichao Sun, Pan Zhou, Shijie Zhang, Yao Wan","submitted_at":"2024-04-22T15:54:53Z","abstract_excerpt":"Recent years have witnessed significant progress in developing deep learning-based models for automated code completion. Although using source code in GitHub has been a common practice for training deep-learning-based models for code completion, it may induce some legal and ethical issues such as copyright infringement. In this paper, we investigate the legal and ethical issues of current neural code completion models by answering the following question: Is my code used to train your neural code completion model? To this end, we tailor a membership inference approach (termed CodeMI) that was o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.14296","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.14296/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.14296","created_at":"2026-07-05T09:04:09.411714+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.14296v2","created_at":"2026-07-05T09:04:09.411714+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.14296","created_at":"2026-07-05T09:04:09.411714+00:00"},{"alias_kind":"pith_short_12","alias_value":"54MPQKZOWCV3","created_at":"2026-07-05T09:04:09.411714+00:00"},{"alias_kind":"pith_short_16","alias_value":"54MPQKZOWCV3WPTY","created_at":"2026-07-05T09:04:09.411714+00:00"},{"alias_kind":"pith_short_8","alias_value":"54MPQKZO","created_at":"2026-07-05T09:04:09.411714+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.26133","citing_title":"Pretraining Data Exposure in Large Language Models: A Survey of Membership Inference, Data Contamination, and Security Implications","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05267","citing_title":"Bridging Generation and Training: A Systematic Review of Quality Issues in LLMs for Code","ref_index":125,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/54MPQKZOWCV3WPTYN3ZBVL7PDO","json":"https://pith.science/pith/54MPQKZOWCV3WPTYN3ZBVL7PDO.json","graph_json":"https://pith.science/api/pith-number/54MPQKZOWCV3WPTYN3ZBVL7PDO/graph.json","events_json":"https://pith.science/api/pith-number/54MPQKZOWCV3WPTYN3ZBVL7PDO/events.json","paper":"https://pith.science/paper/54MPQKZO"},"agent_actions":{"view_html":"https://pith.science/pith/54MPQKZOWCV3WPTYN3ZBVL7PDO","download_json":"https://pith.science/pith/54MPQKZOWCV3WPTYN3ZBVL7PDO.json","view_paper":"https://pith.science/paper/54MPQKZO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.14296&json=true","fetch_graph":"https://pith.science/api/pith-number/54MPQKZOWCV3WPTYN3ZBVL7PDO/graph.json","fetch_events":"https://pith.science/api/pith-number/54MPQKZOWCV3WPTYN3ZBVL7PDO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/54MPQKZOWCV3WPTYN3ZBVL7PDO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/54MPQKZOWCV3WPTYN3ZBVL7PDO/action/storage_attestation","attest_author":"https://pith.science/pith/54MPQKZOWCV3WPTYN3ZBVL7PDO/action/author_attestation","sign_citation":"https://pith.science/pith/54MPQKZOWCV3WPTYN3ZBVL7PDO/action/citation_signature","submit_replication":"https://pith.science/pith/54MPQKZOWCV3WPTYN3ZBVL7PDO/action/replication_record"}},"created_at":"2026-07-05T09:04:09.411714+00:00","updated_at":"2026-07-05T09:04:09.411714+00:00"}