{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:SLXHFLSFYELQ24GGYS72SZTLSB","short_pith_number":"pith:SLXHFLSF","schema_version":"1.0","canonical_sha256":"92ee72ae45c1170d70c6c4bfa9666b906ead48dc75a48ca406cd6ea3816e969f","source":{"kind":"arxiv","id":"2402.12329","version":2},"attestation_state":"computed","paper":{"title":"Query-Based Adversarial Prompt Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Ema Borevkovic, Florian Tram\\`er, Jonathan Hayase, Milad Nasr, Nicholas Carlini","submitted_at":"2024-02-19T18:01:36Z","abstract_excerpt":"Recent work has shown it is possible to construct adversarial examples that cause an aligned language model to emit harmful strings or perform harmful behavior. Existing attacks work either in the white-box setting (with full access to the model weights), or through transferability: the phenomenon that adversarial examples crafted on one model often remain effective on other models. We improve on prior work with a query-based attack that leverages API access to a remote language model to construct adversarial examples that cause the model to emit harmful strings with (much) higher probability "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.12329","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-02-19T18:01:36Z","cross_cats_sorted":["cs.AI","cs.CR","cs.LG"],"title_canon_sha256":"bf1d2de0414c9d254bc5d9076fe1109921d558300100d066ada84fc5479dce20","abstract_canon_sha256":"359c7a34e98323a08b95e4d0ba2904b642dd861c461dd73f3ed602263cf2badf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:45:44.678687Z","signature_b64":"xEcLpEI9cpXM0XW1IDLqWzF3HBHnPzyamD2SCum7OrQ+j5vwuxcZhr6xKrgMBbrG3BXEcutgtHjF9Z5h61vICA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"92ee72ae45c1170d70c6c4bfa9666b906ead48dc75a48ca406cd6ea3816e969f","last_reissued_at":"2026-07-05T09:45:44.678094Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:45:44.678094Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Query-Based Adversarial Prompt Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Ema Borevkovic, Florian Tram\\`er, Jonathan Hayase, Milad Nasr, Nicholas Carlini","submitted_at":"2024-02-19T18:01:36Z","abstract_excerpt":"Recent work has shown it is possible to construct adversarial examples that cause an aligned language model to emit harmful strings or perform harmful behavior. Existing attacks work either in the white-box setting (with full access to the model weights), or through transferability: the phenomenon that adversarial examples crafted on one model often remain effective on other models. We improve on prior work with a query-based attack that leverages API access to a remote language model to construct adversarial examples that cause the model to emit harmful strings with (much) higher probability "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.12329","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.12329/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.12329","created_at":"2026-07-05T09:45:44.678164+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.12329v2","created_at":"2026-07-05T09:45:44.678164+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.12329","created_at":"2026-07-05T09:45:44.678164+00:00"},{"alias_kind":"pith_short_12","alias_value":"SLXHFLSFYELQ","created_at":"2026-07-05T09:45:44.678164+00:00"},{"alias_kind":"pith_short_16","alias_value":"SLXHFLSFYELQ24GG","created_at":"2026-07-05T09:45:44.678164+00:00"},{"alias_kind":"pith_short_8","alias_value":"SLXHFLSF","created_at":"2026-07-05T09:45:44.678164+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2508.20325","citing_title":"GUARD: Guideline Upholding Test through Adaptive Role-play and Jailbreak Diagnostics for LLMs","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2404.01318","citing_title":"JailbreakBench: An Open Robustness Benchmark for Jailbreaking Large Language Models","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2407.04295","citing_title":"Jailbreak Attacks and Defenses Against Large Language Models: A Survey","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20994","citing_title":"Breaking MCP with Function Hijacking Attacks: Novel Threats for Function Calling and Agentic Models","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SLXHFLSFYELQ24GGYS72SZTLSB","json":"https://pith.science/pith/SLXHFLSFYELQ24GGYS72SZTLSB.json","graph_json":"https://pith.science/api/pith-number/SLXHFLSFYELQ24GGYS72SZTLSB/graph.json","events_json":"https://pith.science/api/pith-number/SLXHFLSFYELQ24GGYS72SZTLSB/events.json","paper":"https://pith.science/paper/SLXHFLSF"},"agent_actions":{"view_html":"https://pith.science/pith/SLXHFLSFYELQ24GGYS72SZTLSB","download_json":"https://pith.science/pith/SLXHFLSFYELQ24GGYS72SZTLSB.json","view_paper":"https://pith.science/paper/SLXHFLSF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.12329&json=true","fetch_graph":"https://pith.science/api/pith-number/SLXHFLSFYELQ24GGYS72SZTLSB/graph.json","fetch_events":"https://pith.science/api/pith-number/SLXHFLSFYELQ24GGYS72SZTLSB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SLXHFLSFYELQ24GGYS72SZTLSB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SLXHFLSFYELQ24GGYS72SZTLSB/action/storage_attestation","attest_author":"https://pith.science/pith/SLXHFLSFYELQ24GGYS72SZTLSB/action/author_attestation","sign_citation":"https://pith.science/pith/SLXHFLSFYELQ24GGYS72SZTLSB/action/citation_signature","submit_replication":"https://pith.science/pith/SLXHFLSFYELQ24GGYS72SZTLSB/action/replication_record"}},"created_at":"2026-07-05T09:45:44.678164+00:00","updated_at":"2026-07-05T09:45:44.678164+00:00"}