{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YMCOCF737EA3ATSSBHD22GPS5P","short_pith_number":"pith:YMCOCF73","schema_version":"1.0","canonical_sha256":"c304e117fbf901b04e5209c7ad19f2ebf9932b962ef4745476e981db2d81f730","source":{"kind":"arxiv","id":"2411.16721","version":3},"attestation_state":"computed","paper":{"title":"Steering Away from Harm: An Adaptive Approach to Defending Vision Language Model Against Jailbreaks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Gang Wang, Han Wang, Huan Zhang","submitted_at":"2024-11-23T02:17:17Z","abstract_excerpt":"Vision Language Models (VLMs) can produce unintended and harmful content when exposed to adversarial attacks, particularly because their vision capabilities create new vulnerabilities. Existing defenses, such as input preprocessing, adversarial training, and response evaluation-based methods, are often impractical for real-world deployment due to their high costs. To address this challenge, we propose ASTRA, an efficient and effective defense by adaptively steering models away from adversarial feature directions to resist VLM attacks. Our key procedures involve finding transferable steering ve"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.16721","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-11-23T02:17:17Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"6afe68746175137603f574df0df48f06388c1450bf24ef6d5eeea51cae21e58d","abstract_canon_sha256":"7b2cceebcffbba4bec33bc23e5fa04b087379c88bb47a191b47df83d0590c6e6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:57:30.014020Z","signature_b64":"FuBTACwMzRQDQfYge4/n9J3mH2zXwNMQ6j94OFr06JgYAiwpOgTgUeLqi3zPvya9FoODoLuBPKQA/PNzuN4VDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c304e117fbf901b04e5209c7ad19f2ebf9932b962ef4745476e981db2d81f730","last_reissued_at":"2026-07-05T10:57:30.013483Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:57:30.013483Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Steering Away from Harm: An Adaptive Approach to Defending Vision Language Model Against Jailbreaks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Gang Wang, Han Wang, Huan Zhang","submitted_at":"2024-11-23T02:17:17Z","abstract_excerpt":"Vision Language Models (VLMs) can produce unintended and harmful content when exposed to adversarial attacks, particularly because their vision capabilities create new vulnerabilities. Existing defenses, such as input preprocessing, adversarial training, and response evaluation-based methods, are often impractical for real-world deployment due to their high costs. To address this challenge, we propose ASTRA, an efficient and effective defense by adaptively steering models away from adversarial feature directions to resist VLM attacks. Our key procedures involve finding transferable steering ve"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.16721","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.16721/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.16721","created_at":"2026-07-05T10:57:30.013550+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.16721v3","created_at":"2026-07-05T10:57:30.013550+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.16721","created_at":"2026-07-05T10:57:30.013550+00:00"},{"alias_kind":"pith_short_12","alias_value":"YMCOCF737EA3","created_at":"2026-07-05T10:57:30.013550+00:00"},{"alias_kind":"pith_short_16","alias_value":"YMCOCF737EA3ATSS","created_at":"2026-07-05T10:57:30.013550+00:00"},{"alias_kind":"pith_short_8","alias_value":"YMCOCF73","created_at":"2026-07-05T10:57:30.013550+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.24519","citing_title":"AMIA: Automatic Masking and Joint Intention Analysis Makes LVLMs Robust Jailbreak Defenders","ref_index":26,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YMCOCF737EA3ATSSBHD22GPS5P","json":"https://pith.science/pith/YMCOCF737EA3ATSSBHD22GPS5P.json","graph_json":"https://pith.science/api/pith-number/YMCOCF737EA3ATSSBHD22GPS5P/graph.json","events_json":"https://pith.science/api/pith-number/YMCOCF737EA3ATSSBHD22GPS5P/events.json","paper":"https://pith.science/paper/YMCOCF73"},"agent_actions":{"view_html":"https://pith.science/pith/YMCOCF737EA3ATSSBHD22GPS5P","download_json":"https://pith.science/pith/YMCOCF737EA3ATSSBHD22GPS5P.json","view_paper":"https://pith.science/paper/YMCOCF73","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.16721&json=true","fetch_graph":"https://pith.science/api/pith-number/YMCOCF737EA3ATSSBHD22GPS5P/graph.json","fetch_events":"https://pith.science/api/pith-number/YMCOCF737EA3ATSSBHD22GPS5P/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YMCOCF737EA3ATSSBHD22GPS5P/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YMCOCF737EA3ATSSBHD22GPS5P/action/storage_attestation","attest_author":"https://pith.science/pith/YMCOCF737EA3ATSSBHD22GPS5P/action/author_attestation","sign_citation":"https://pith.science/pith/YMCOCF737EA3ATSSBHD22GPS5P/action/citation_signature","submit_replication":"https://pith.science/pith/YMCOCF737EA3ATSSBHD22GPS5P/action/replication_record"}},"created_at":"2026-07-05T10:57:30.013550+00:00","updated_at":"2026-07-05T10:57:30.013550+00:00"}