{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:WEEPXQQXNXQA56FSZ4EZPSJFER","short_pith_number":"pith:WEEPXQQX","schema_version":"1.0","canonical_sha256":"b108fbc2176de00ef8b2cf0997c9252462b7826f22f161a68da578342a7f77cc","source":{"kind":"arxiv","id":"2410.08858","version":2},"attestation_state":"computed","paper":{"title":"Decoding Secret Memorization in Code LLMs Through Token-Level Characterization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SE"],"primary_cat":"cs.CR","authors_text":"Chong Wang, Guoai Xu, Guosheng Xu, Haoyu Wang, Kailong Wang, Yuqing Nie","submitted_at":"2024-10-11T14:39:24Z","abstract_excerpt":"Code Large Language Models (LLMs) have demonstrated remarkable capabilities in generating, understanding, and manipulating programming code. However, their training process inadvertently leads to the memorization of sensitive information, posing severe privacy risks. Existing studies on memorization in LLMs primarily rely on prompt engineering techniques, which suffer from limitations such as widespread hallucination and inefficient extraction of the target sensitive information. In this paper, we present a novel approach to characterize real and fake secrets generated by Code LLMs based on to"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.08858","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2024-10-11T14:39:24Z","cross_cats_sorted":["cs.SE"],"title_canon_sha256":"bf2ce352ad88d489a4db9010e069e2b35e6d77e2e339886a548a7ce9c9d614ea","abstract_canon_sha256":"eec5b0bfb640e28ad60b2bca38d81482be30e68abe749a345525013eae661549"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:51:26.280892Z","signature_b64":"sS9yY5Rb36onNrVxbxWwe+f4Ud+i7G6WfEwuHDy1LUT7Mz2iTYApvbDRDeldlXzCIPi9KzPnHr7TbEnQnBZvBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b108fbc2176de00ef8b2cf0997c9252462b7826f22f161a68da578342a7f77cc","last_reissued_at":"2026-07-05T10:51:26.280403Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:51:26.280403Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Decoding Secret Memorization in Code LLMs Through Token-Level Characterization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SE"],"primary_cat":"cs.CR","authors_text":"Chong Wang, Guoai Xu, Guosheng Xu, Haoyu Wang, Kailong Wang, Yuqing Nie","submitted_at":"2024-10-11T14:39:24Z","abstract_excerpt":"Code Large Language Models (LLMs) have demonstrated remarkable capabilities in generating, understanding, and manipulating programming code. However, their training process inadvertently leads to the memorization of sensitive information, posing severe privacy risks. Existing studies on memorization in LLMs primarily rely on prompt engineering techniques, which suffer from limitations such as widespread hallucination and inefficient extraction of the target sensitive information. In this paper, we present a novel approach to characterize real and fake secrets generated by Code LLMs based on to"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.08858","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.08858/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.08858","created_at":"2026-07-05T10:51:26.280458+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.08858v2","created_at":"2026-07-05T10:51:26.280458+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.08858","created_at":"2026-07-05T10:51:26.280458+00:00"},{"alias_kind":"pith_short_12","alias_value":"WEEPXQQXNXQA","created_at":"2026-07-05T10:51:26.280458+00:00"},{"alias_kind":"pith_short_16","alias_value":"WEEPXQQXNXQA56FS","created_at":"2026-07-05T10:51:26.280458+00:00"},{"alias_kind":"pith_short_8","alias_value":"WEEPXQQX","created_at":"2026-07-05T10:51:26.280458+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.26133","citing_title":"Pretraining Data Exposure in Large Language Models: A Survey of Membership Inference, Data Contamination, and Security Implications","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05267","citing_title":"Bridging Generation and Training: A Systematic Review of Quality Issues in LLMs for Code","ref_index":90,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WEEPXQQXNXQA56FSZ4EZPSJFER","json":"https://pith.science/pith/WEEPXQQXNXQA56FSZ4EZPSJFER.json","graph_json":"https://pith.science/api/pith-number/WEEPXQQXNXQA56FSZ4EZPSJFER/graph.json","events_json":"https://pith.science/api/pith-number/WEEPXQQXNXQA56FSZ4EZPSJFER/events.json","paper":"https://pith.science/paper/WEEPXQQX"},"agent_actions":{"view_html":"https://pith.science/pith/WEEPXQQXNXQA56FSZ4EZPSJFER","download_json":"https://pith.science/pith/WEEPXQQXNXQA56FSZ4EZPSJFER.json","view_paper":"https://pith.science/paper/WEEPXQQX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.08858&json=true","fetch_graph":"https://pith.science/api/pith-number/WEEPXQQXNXQA56FSZ4EZPSJFER/graph.json","fetch_events":"https://pith.science/api/pith-number/WEEPXQQXNXQA56FSZ4EZPSJFER/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WEEPXQQXNXQA56FSZ4EZPSJFER/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WEEPXQQXNXQA56FSZ4EZPSJFER/action/storage_attestation","attest_author":"https://pith.science/pith/WEEPXQQXNXQA56FSZ4EZPSJFER/action/author_attestation","sign_citation":"https://pith.science/pith/WEEPXQQXNXQA56FSZ4EZPSJFER/action/citation_signature","submit_replication":"https://pith.science/pith/WEEPXQQXNXQA56FSZ4EZPSJFER/action/replication_record"}},"created_at":"2026-07-05T10:51:26.280458+00:00","updated_at":"2026-07-05T10:51:26.280458+00:00"}