{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:UIN3SU7DJRTOYYJNBF3DDABEYR","short_pith_number":"pith:UIN3SU7D","schema_version":"1.0","canonical_sha256":"a21bb953e34c66ec612d0976318024c4458e1c9e05680dff74077d61e9419971","source":{"kind":"arxiv","id":"2410.09047","version":1},"attestation_state":"computed","paper":{"title":"Unraveling and Mitigating Safety Alignment Degradation of Vision-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chao Shang, Jie Ma, Ling Liu, Lluis Marquez, Miguel Ballesteros, Neha Anna John, Nikolaos Pappas, Qin Liu, Srikanth Doss, Yassine Benajiba","submitted_at":"2024-10-11T17:59:31Z","abstract_excerpt":"The safety alignment ability of Vision-Language Models (VLMs) is prone to be degraded by the integration of the vision module compared to its LLM backbone. We investigate this phenomenon, dubbed as ''safety alignment degradation'' in this paper, and show that the challenge arises from the representation gap that emerges when introducing vision modality to VLMs. In particular, we show that the representations of multi-modal inputs shift away from that of text-only inputs which represent the distribution that the LLM backbone is optimized for. At the same time, the safety alignment capabilities,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.09047","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-11T17:59:31Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"9a0c84317367beb74ad199ff7dd94deffbd2fed73cef1b7491c2e4d222545438","abstract_canon_sha256":"438544a4846015e77b60c0e71b2ed27707928d1206ad1bf3cb14681d696f63e9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:19:21.185794Z","signature_b64":"0/8CEfXp+uVYvwPVAH5IGt/xyZUZ5Cqn29IKzK3iXMl54EmTz8qWnFdJO3KL5vD9VW+bCrEJCKtpPmATSnQyBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a21bb953e34c66ec612d0976318024c4458e1c9e05680dff74077d61e9419971","last_reissued_at":"2026-07-05T09:19:21.185377Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:19:21.185377Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Unraveling and Mitigating Safety Alignment Degradation of Vision-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chao Shang, Jie Ma, Ling Liu, Lluis Marquez, Miguel Ballesteros, Neha Anna John, Nikolaos Pappas, Qin Liu, Srikanth Doss, Yassine Benajiba","submitted_at":"2024-10-11T17:59:31Z","abstract_excerpt":"The safety alignment ability of Vision-Language Models (VLMs) is prone to be degraded by the integration of the vision module compared to its LLM backbone. We investigate this phenomenon, dubbed as ''safety alignment degradation'' in this paper, and show that the challenge arises from the representation gap that emerges when introducing vision modality to VLMs. In particular, we show that the representations of multi-modal inputs shift away from that of text-only inputs which represent the distribution that the LLM backbone is optimized for. At the same time, the safety alignment capabilities,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.09047","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.09047/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.09047","created_at":"2026-07-05T09:19:21.185434+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.09047v1","created_at":"2026-07-05T09:19:21.185434+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.09047","created_at":"2026-07-05T09:19:21.185434+00:00"},{"alias_kind":"pith_short_12","alias_value":"UIN3SU7DJRTO","created_at":"2026-07-05T09:19:21.185434+00:00"},{"alias_kind":"pith_short_16","alias_value":"UIN3SU7DJRTOYYJN","created_at":"2026-07-05T09:19:21.185434+00:00"},{"alias_kind":"pith_short_8","alias_value":"UIN3SU7D","created_at":"2026-07-05T09:19:21.185434+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UIN3SU7DJRTOYYJNBF3DDABEYR","json":"https://pith.science/pith/UIN3SU7DJRTOYYJNBF3DDABEYR.json","graph_json":"https://pith.science/api/pith-number/UIN3SU7DJRTOYYJNBF3DDABEYR/graph.json","events_json":"https://pith.science/api/pith-number/UIN3SU7DJRTOYYJNBF3DDABEYR/events.json","paper":"https://pith.science/paper/UIN3SU7D"},"agent_actions":{"view_html":"https://pith.science/pith/UIN3SU7DJRTOYYJNBF3DDABEYR","download_json":"https://pith.science/pith/UIN3SU7DJRTOYYJNBF3DDABEYR.json","view_paper":"https://pith.science/paper/UIN3SU7D","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.09047&json=true","fetch_graph":"https://pith.science/api/pith-number/UIN3SU7DJRTOYYJNBF3DDABEYR/graph.json","fetch_events":"https://pith.science/api/pith-number/UIN3SU7DJRTOYYJNBF3DDABEYR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UIN3SU7DJRTOYYJNBF3DDABEYR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UIN3SU7DJRTOYYJNBF3DDABEYR/action/storage_attestation","attest_author":"https://pith.science/pith/UIN3SU7DJRTOYYJNBF3DDABEYR/action/author_attestation","sign_citation":"https://pith.science/pith/UIN3SU7DJRTOYYJNBF3DDABEYR/action/citation_signature","submit_replication":"https://pith.science/pith/UIN3SU7DJRTOYYJNBF3DDABEYR/action/replication_record"}},"created_at":"2026-07-05T09:19:21.185434+00:00","updated_at":"2026-07-05T09:19:21.185434+00:00"}