{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QTE7Z6D4QT3STV2EKATX4FK7DA","short_pith_number":"pith:QTE7Z6D4","schema_version":"1.0","canonical_sha256":"84c9fcf87c84f729d74450277e155f1806ea4ba5d3fdba0e5cfd5ff8de5cb929","source":{"kind":"arxiv","id":"2409.13853","version":1},"attestation_state":"computed","paper":{"title":"Unlocking Memorization in Large Language Models with Dynamic Soft Prompting","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Cao Xiao, Feng Zheng, Jackson Taylor, Runxue Bao, Shangqian Gao, Weiwen Jiang, Yanfu Zhang, Yawen Wu, Zhepeng Wang","submitted_at":"2024-09-20T18:56:32Z","abstract_excerpt":"Pretrained large language models (LLMs) have revolutionized natural language processing (NLP) tasks such as summarization, question answering, and translation. However, LLMs pose significant security risks due to their tendency to memorize training data, leading to potential privacy breaches and copyright infringement. Accurate measurement of this memorization is essential to evaluate and mitigate these potential risks. However, previous attempts to characterize memorization are constrained by either using prefixes only or by prepending a constant soft prompt to the prefixes, which cannot reac"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.13853","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-09-20T18:56:32Z","cross_cats_sorted":["cs.AI","cs.CR","cs.LG"],"title_canon_sha256":"5cd297bfb4fdb4e1457a53953748ba2b83a40a03d6f9eb7a6d3d8f3f8cd1fcea","abstract_canon_sha256":"89a47dc6f856b9e6ae56d0615c1f1ac251a72cf081c97f8fbff769df41adfacf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:10:07.709378Z","signature_b64":"ZB8Of3U5eB/+Q6P9LB3c850PLFOBQ36Riwsk8TIBkEy86DiGqY0of22fZrZaps8OOZLAIVcT0cZd9sbwaxPxCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"84c9fcf87c84f729d74450277e155f1806ea4ba5d3fdba0e5cfd5ff8de5cb929","last_reissued_at":"2026-07-05T09:10:07.708887Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:10:07.708887Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Unlocking Memorization in Large Language Models with Dynamic Soft Prompting","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Cao Xiao, Feng Zheng, Jackson Taylor, Runxue Bao, Shangqian Gao, Weiwen Jiang, Yanfu Zhang, Yawen Wu, Zhepeng Wang","submitted_at":"2024-09-20T18:56:32Z","abstract_excerpt":"Pretrained large language models (LLMs) have revolutionized natural language processing (NLP) tasks such as summarization, question answering, and translation. However, LLMs pose significant security risks due to their tendency to memorize training data, leading to potential privacy breaches and copyright infringement. Accurate measurement of this memorization is essential to evaluate and mitigate these potential risks. However, previous attempts to characterize memorization are constrained by either using prefixes only or by prepending a constant soft prompt to the prefixes, which cannot reac"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.13853","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.13853/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.13853","created_at":"2026-07-05T09:10:07.708946+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.13853v1","created_at":"2026-07-05T09:10:07.708946+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.13853","created_at":"2026-07-05T09:10:07.708946+00:00"},{"alias_kind":"pith_short_12","alias_value":"QTE7Z6D4QT3S","created_at":"2026-07-05T09:10:07.708946+00:00"},{"alias_kind":"pith_short_16","alias_value":"QTE7Z6D4QT3STV2E","created_at":"2026-07-05T09:10:07.708946+00:00"},{"alias_kind":"pith_short_8","alias_value":"QTE7Z6D4","created_at":"2026-07-05T09:10:07.708946+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2412.20760","citing_title":"Attributing Culture-Conditioned Generations to Pretraining Corpora","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QTE7Z6D4QT3STV2EKATX4FK7DA","json":"https://pith.science/pith/QTE7Z6D4QT3STV2EKATX4FK7DA.json","graph_json":"https://pith.science/api/pith-number/QTE7Z6D4QT3STV2EKATX4FK7DA/graph.json","events_json":"https://pith.science/api/pith-number/QTE7Z6D4QT3STV2EKATX4FK7DA/events.json","paper":"https://pith.science/paper/QTE7Z6D4"},"agent_actions":{"view_html":"https://pith.science/pith/QTE7Z6D4QT3STV2EKATX4FK7DA","download_json":"https://pith.science/pith/QTE7Z6D4QT3STV2EKATX4FK7DA.json","view_paper":"https://pith.science/paper/QTE7Z6D4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.13853&json=true","fetch_graph":"https://pith.science/api/pith-number/QTE7Z6D4QT3STV2EKATX4FK7DA/graph.json","fetch_events":"https://pith.science/api/pith-number/QTE7Z6D4QT3STV2EKATX4FK7DA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QTE7Z6D4QT3STV2EKATX4FK7DA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QTE7Z6D4QT3STV2EKATX4FK7DA/action/storage_attestation","attest_author":"https://pith.science/pith/QTE7Z6D4QT3STV2EKATX4FK7DA/action/author_attestation","sign_citation":"https://pith.science/pith/QTE7Z6D4QT3STV2EKATX4FK7DA/action/citation_signature","submit_replication":"https://pith.science/pith/QTE7Z6D4QT3STV2EKATX4FK7DA/action/replication_record"}},"created_at":"2026-07-05T09:10:07.708946+00:00","updated_at":"2026-07-05T09:10:07.708946+00:00"}