{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:UQFNFVXSCA6CUO3CILMD6MSQMA","short_pith_number":"pith:UQFNFVXS","schema_version":"1.0","canonical_sha256":"a40ad2d6f2103c2a3b6242d83f32506025517acfb4b6d82b1050bf81c0d7819a","source":{"kind":"arxiv","id":"2403.04801","version":3},"attestation_state":"computed","paper":{"title":"Alpaca against Vicuna: Using LLMs to Uncover Memorization of LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Aly M. Kassem, Hyunwoo Kim, Niloofar Mireshghallah, Omar Mahmoud, Santu Rana, Sherif Saad, Yejin Choi, Yulia Tsvetkov","submitted_at":"2024-03-05T19:32:01Z","abstract_excerpt":"In this paper, we introduce a black-box prompt optimization method that uses an attacker LLM agent to uncover higher levels of memorization in a victim agent, compared to what is revealed by prompting the target model with the training data directly, which is the dominant approach of quantifying memorization in LLMs. We use an iterative rejection-sampling optimization process to find instruction-based prompts with two main characteristics: (1) minimal overlap with the training data to avoid presenting the solution directly to the model, and (2) maximal overlap between the victim model's output"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.04801","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-03-05T19:32:01Z","cross_cats_sorted":[],"title_canon_sha256":"6dcc17d0f6001d119ae5e917640333e9ad91889ab87b2f26f0c377573123f614","abstract_canon_sha256":"5c76ce5fd8cb5045898ffc326a855438a87f5fe13792a00f28f4d0ce9081c6c1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:11:17.467725Z","signature_b64":"3lrPW244JYeJgXWtRycVQ+CZqL5/jneIVTdJRqU7uVqJzFY3P/il7lo0KGUaUBGoikWdSuCWw6C69XTZIjIeBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a40ad2d6f2103c2a3b6242d83f32506025517acfb4b6d82b1050bf81c0d7819a","last_reissued_at":"2026-07-05T10:11:17.467216Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:11:17.467216Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Alpaca against Vicuna: Using LLMs to Uncover Memorization of LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Aly M. Kassem, Hyunwoo Kim, Niloofar Mireshghallah, Omar Mahmoud, Santu Rana, Sherif Saad, Yejin Choi, Yulia Tsvetkov","submitted_at":"2024-03-05T19:32:01Z","abstract_excerpt":"In this paper, we introduce a black-box prompt optimization method that uses an attacker LLM agent to uncover higher levels of memorization in a victim agent, compared to what is revealed by prompting the target model with the training data directly, which is the dominant approach of quantifying memorization in LLMs. We use an iterative rejection-sampling optimization process to find instruction-based prompts with two main characteristics: (1) minimal overlap with the training data to avoid presenting the solution directly to the model, and (2) maximal overlap between the victim model's output"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.04801","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.04801/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.04801","created_at":"2026-07-05T10:11:17.467276+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.04801v3","created_at":"2026-07-05T10:11:17.467276+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.04801","created_at":"2026-07-05T10:11:17.467276+00:00"},{"alias_kind":"pith_short_12","alias_value":"UQFNFVXSCA6C","created_at":"2026-07-05T10:11:17.467276+00:00"},{"alias_kind":"pith_short_16","alias_value":"UQFNFVXSCA6CUO3C","created_at":"2026-07-05T10:11:17.467276+00:00"},{"alias_kind":"pith_short_8","alias_value":"UQFNFVXS","created_at":"2026-07-05T10:11:17.467276+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.05206","citing_title":"Safety at Scale: A Comprehensive Survey of Large Model and Agent Safety","ref_index":212,"is_internal_anchor":false},{"citing_arxiv_id":"2406.08464","citing_title":"Magpie: Alignment Data Synthesis from Scratch by Prompting Aligned LLMs with Nothing","ref_index":115,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18697","citing_title":"Beyond Indistinguishability: Measuring Extraction Risk in LLM APIs","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UQFNFVXSCA6CUO3CILMD6MSQMA","json":"https://pith.science/pith/UQFNFVXSCA6CUO3CILMD6MSQMA.json","graph_json":"https://pith.science/api/pith-number/UQFNFVXSCA6CUO3CILMD6MSQMA/graph.json","events_json":"https://pith.science/api/pith-number/UQFNFVXSCA6CUO3CILMD6MSQMA/events.json","paper":"https://pith.science/paper/UQFNFVXS"},"agent_actions":{"view_html":"https://pith.science/pith/UQFNFVXSCA6CUO3CILMD6MSQMA","download_json":"https://pith.science/pith/UQFNFVXSCA6CUO3CILMD6MSQMA.json","view_paper":"https://pith.science/paper/UQFNFVXS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.04801&json=true","fetch_graph":"https://pith.science/api/pith-number/UQFNFVXSCA6CUO3CILMD6MSQMA/graph.json","fetch_events":"https://pith.science/api/pith-number/UQFNFVXSCA6CUO3CILMD6MSQMA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UQFNFVXSCA6CUO3CILMD6MSQMA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UQFNFVXSCA6CUO3CILMD6MSQMA/action/storage_attestation","attest_author":"https://pith.science/pith/UQFNFVXSCA6CUO3CILMD6MSQMA/action/author_attestation","sign_citation":"https://pith.science/pith/UQFNFVXSCA6CUO3CILMD6MSQMA/action/citation_signature","submit_replication":"https://pith.science/pith/UQFNFVXSCA6CUO3CILMD6MSQMA/action/replication_record"}},"created_at":"2026-07-05T10:11:17.467276+00:00","updated_at":"2026-07-05T10:11:17.467276+00:00"}