{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:WDIZN3N7RSIXUITCRS7INFXGCK","short_pith_number":"pith:WDIZN3N7","schema_version":"1.0","canonical_sha256":"b0d196edbf8c917a22628cbe8696e61289a42d8461dcdd34aff1e170d1384d31","source":{"kind":"arxiv","id":"2505.24227","version":1},"attestation_state":"computed","paper":{"title":"Light as Deception: GPT-driven Natural Relighting Against Vision-Language Pre-training Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CR"],"primary_cat":"cs.CV","authors_text":"Di Lin, Jie Zhang, Qing Guo, Tao Xiang, Xiao Lv, Ying Yang","submitted_at":"2025-05-30T05:30:02Z","abstract_excerpt":"While adversarial attacks on vision-and-language pretraining (VLP) models have been explored, generating natural adversarial samples crafted through realistic and semantically meaningful perturbations remains an open challenge. Existing methods, primarily designed for classification tasks, struggle when adapted to VLP models due to their restricted optimization spaces, leading to ineffective attacks or unnatural artifacts. To address this, we propose \\textbf{LightD}, a novel framework that generates natural adversarial samples for VLP models via semantically guided relighting. Specifically, Li"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.24227","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-05-30T05:30:02Z","cross_cats_sorted":["cs.CR"],"title_canon_sha256":"b7757d9a2eb4e61107ceb02d3615c59a3767337accf1db7c216e58a29caac2dc","abstract_canon_sha256":"9310b554dbf18bf34736e1ba876a91406ab38324c16d8b8db75789d8c1df9578"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:12:44.600462Z","signature_b64":"vdRAyUd5xtI3poAV8Ad+er1OhtB1eLighlY9MhwUI7HS8E04KtG+EvUnHo4kZAUlwzI2dqo9sBnenpku0xykDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b0d196edbf8c917a22628cbe8696e61289a42d8461dcdd34aff1e170d1384d31","last_reissued_at":"2026-07-05T11:12:44.599874Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:12:44.599874Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Light as Deception: GPT-driven Natural Relighting Against Vision-Language Pre-training Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CR"],"primary_cat":"cs.CV","authors_text":"Di Lin, Jie Zhang, Qing Guo, Tao Xiang, Xiao Lv, Ying Yang","submitted_at":"2025-05-30T05:30:02Z","abstract_excerpt":"While adversarial attacks on vision-and-language pretraining (VLP) models have been explored, generating natural adversarial samples crafted through realistic and semantically meaningful perturbations remains an open challenge. Existing methods, primarily designed for classification tasks, struggle when adapted to VLP models due to their restricted optimization spaces, leading to ineffective attacks or unnatural artifacts. To address this, we propose \\textbf{LightD}, a novel framework that generates natural adversarial samples for VLP models via semantically guided relighting. Specifically, Li"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.24227","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.24227/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.24227","created_at":"2026-07-05T11:12:44.599946+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.24227v1","created_at":"2026-07-05T11:12:44.599946+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.24227","created_at":"2026-07-05T11:12:44.599946+00:00"},{"alias_kind":"pith_short_12","alias_value":"WDIZN3N7RSIX","created_at":"2026-07-05T11:12:44.599946+00:00"},{"alias_kind":"pith_short_16","alias_value":"WDIZN3N7RSIXUITC","created_at":"2026-07-05T11:12:44.599946+00:00"},{"alias_kind":"pith_short_8","alias_value":"WDIZN3N7","created_at":"2026-07-05T11:12:44.599946+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.12833","citing_title":"Challenging Vision-Language Models with Physically Deployable Multimodal Semantic Lighting Attacks","ref_index":43,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WDIZN3N7RSIXUITCRS7INFXGCK","json":"https://pith.science/pith/WDIZN3N7RSIXUITCRS7INFXGCK.json","graph_json":"https://pith.science/api/pith-number/WDIZN3N7RSIXUITCRS7INFXGCK/graph.json","events_json":"https://pith.science/api/pith-number/WDIZN3N7RSIXUITCRS7INFXGCK/events.json","paper":"https://pith.science/paper/WDIZN3N7"},"agent_actions":{"view_html":"https://pith.science/pith/WDIZN3N7RSIXUITCRS7INFXGCK","download_json":"https://pith.science/pith/WDIZN3N7RSIXUITCRS7INFXGCK.json","view_paper":"https://pith.science/paper/WDIZN3N7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.24227&json=true","fetch_graph":"https://pith.science/api/pith-number/WDIZN3N7RSIXUITCRS7INFXGCK/graph.json","fetch_events":"https://pith.science/api/pith-number/WDIZN3N7RSIXUITCRS7INFXGCK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WDIZN3N7RSIXUITCRS7INFXGCK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WDIZN3N7RSIXUITCRS7INFXGCK/action/storage_attestation","attest_author":"https://pith.science/pith/WDIZN3N7RSIXUITCRS7INFXGCK/action/author_attestation","sign_citation":"https://pith.science/pith/WDIZN3N7RSIXUITCRS7INFXGCK/action/citation_signature","submit_replication":"https://pith.science/pith/WDIZN3N7RSIXUITCRS7INFXGCK/action/replication_record"}},"created_at":"2026-07-05T11:12:44.599946+00:00","updated_at":"2026-07-05T11:12:44.599946+00:00"}