{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:SESJ7F3VU5ZDYVAH2FHPMVINYP","short_pith_number":"pith:SESJ7F3V","schema_version":"1.0","canonical_sha256":"91249f9775a7723c5407d14ef6550dc3fdd31925fbddf06fa4ddae8467f3d663","source":{"kind":"arxiv","id":"2407.09292","version":2},"attestation_state":"computed","paper":{"title":"Counterfactual Explainable Incremental Prompt Attack Analysis on Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CR","authors_text":"Chong Zhang, Dong Shu, Mingyu Jin, Tianle Chen, Yongfeng Zhang","submitted_at":"2024-07-12T14:26:14Z","abstract_excerpt":"This study sheds light on the imperative need to bolster safety and privacy measures in large language models (LLMs), such as GPT-4 and LLaMA-2, by identifying and mitigating their vulnerabilities through explainable analysis of prompt attacks. We propose Counterfactual Explainable Incremental Prompt Attack (CEIPA), a novel technique where we guide prompts in a specific manner to quantitatively measure attack effectiveness and explore the embedded defense mechanisms in these models. Our approach is distinctive for its capacity to elucidate the reasons behind the generation of harmful responses"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.09292","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2024-07-12T14:26:14Z","cross_cats_sorted":[],"title_canon_sha256":"aad7bb295d2b92bb46956c3de203a1ebd428ea451b0a579ccf662f1e5497527c","abstract_canon_sha256":"0f6833ba11b804ed6f9b6889b7bd18241906dccf0500abeed5d1cebdc8842699"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:44:56.501523Z","signature_b64":"i2lUyQAuxX0gXfs2IyFVMuoUb/lANxZS0ZYjVpw8hT+gfdxA/KJqWF1yVE3j2Gmy/kY0XVk9Wxc96iPlPXqbCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"91249f9775a7723c5407d14ef6550dc3fdd31925fbddf06fa4ddae8467f3d663","last_reissued_at":"2026-07-05T08:44:56.501071Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:44:56.501071Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Counterfactual Explainable Incremental Prompt Attack Analysis on Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CR","authors_text":"Chong Zhang, Dong Shu, Mingyu Jin, Tianle Chen, Yongfeng Zhang","submitted_at":"2024-07-12T14:26:14Z","abstract_excerpt":"This study sheds light on the imperative need to bolster safety and privacy measures in large language models (LLMs), such as GPT-4 and LLaMA-2, by identifying and mitigating their vulnerabilities through explainable analysis of prompt attacks. We propose Counterfactual Explainable Incremental Prompt Attack (CEIPA), a novel technique where we guide prompts in a specific manner to quantitatively measure attack effectiveness and explore the embedded defense mechanisms in these models. Our approach is distinctive for its capacity to elucidate the reasons behind the generation of harmful responses"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.09292","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.09292/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.09292","created_at":"2026-07-05T08:44:56.501129+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.09292v2","created_at":"2026-07-05T08:44:56.501129+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.09292","created_at":"2026-07-05T08:44:56.501129+00:00"},{"alias_kind":"pith_short_12","alias_value":"SESJ7F3VU5ZD","created_at":"2026-07-05T08:44:56.501129+00:00"},{"alias_kind":"pith_short_16","alias_value":"SESJ7F3VU5ZDYVAH","created_at":"2026-07-05T08:44:56.501129+00:00"},{"alias_kind":"pith_short_8","alias_value":"SESJ7F3V","created_at":"2026-07-05T08:44:56.501129+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.08000","citing_title":"AntiDote: Bi-level Adversarial Training for Tamper-Resistant LLMs","ref_index":72,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SESJ7F3VU5ZDYVAH2FHPMVINYP","json":"https://pith.science/pith/SESJ7F3VU5ZDYVAH2FHPMVINYP.json","graph_json":"https://pith.science/api/pith-number/SESJ7F3VU5ZDYVAH2FHPMVINYP/graph.json","events_json":"https://pith.science/api/pith-number/SESJ7F3VU5ZDYVAH2FHPMVINYP/events.json","paper":"https://pith.science/paper/SESJ7F3V"},"agent_actions":{"view_html":"https://pith.science/pith/SESJ7F3VU5ZDYVAH2FHPMVINYP","download_json":"https://pith.science/pith/SESJ7F3VU5ZDYVAH2FHPMVINYP.json","view_paper":"https://pith.science/paper/SESJ7F3V","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.09292&json=true","fetch_graph":"https://pith.science/api/pith-number/SESJ7F3VU5ZDYVAH2FHPMVINYP/graph.json","fetch_events":"https://pith.science/api/pith-number/SESJ7F3VU5ZDYVAH2FHPMVINYP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SESJ7F3VU5ZDYVAH2FHPMVINYP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SESJ7F3VU5ZDYVAH2FHPMVINYP/action/storage_attestation","attest_author":"https://pith.science/pith/SESJ7F3VU5ZDYVAH2FHPMVINYP/action/author_attestation","sign_citation":"https://pith.science/pith/SESJ7F3VU5ZDYVAH2FHPMVINYP/action/citation_signature","submit_replication":"https://pith.science/pith/SESJ7F3VU5ZDYVAH2FHPMVINYP/action/replication_record"}},"created_at":"2026-07-05T08:44:56.501129+00:00","updated_at":"2026-07-05T08:44:56.501129+00:00"}