{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3W3C3MKAGE3XCWVU4G6XNQ4K7U","short_pith_number":"pith:3W3C3MKA","schema_version":"1.0","canonical_sha256":"ddb62db1403137715ab4e1bd76c38afd39cbe675a9b932cec4298bb55291dc38","source":{"kind":"arxiv","id":"2411.10414","version":1},"attestation_state":"computed","paper":{"title":"Llama Guard 3 Vision: Safeguarding Human-AI Image Understanding Conversations","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Eric Smith, Hongyuan Zhan, Javier Rando, Jianfeng Chi, Kartikeya Upasani, Kate Plawiak, Mahesh Pasupuleti, Ujjwal Karn, Yiming Zhang, Zacharie Delpierre Coudert","submitted_at":"2024-11-15T18:34:07Z","abstract_excerpt":"We introduce Llama Guard 3 Vision, a multimodal LLM-based safeguard for human-AI conversations that involves image understanding: it can be used to safeguard content for both multimodal LLM inputs (prompt classification) and outputs (response classification). Unlike the previous text-only Llama Guard versions (Inan et al., 2023; Llama Team, 2024b,a), it is specifically designed to support image reasoning use cases and is optimized to detect harmful multimodal (text and image) prompts and text responses to these prompts. Llama Guard 3 Vision is fine-tuned on Llama 3.2-Vision and demonstrates st"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.10414","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-11-15T18:34:07Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"9c9282082e3cfcb8209cb17ed290b735576729589385df07cf7a3b57c037bd05","abstract_canon_sha256":"5f744885cb93ce50ec4859d15bc3fe1bc37f695973ffd523ab67f0567e2a7804"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:36:05.055349Z","signature_b64":"L9es8wUqLlAMtD6+wc3S+4ovFZZ120hHRl96I+Bg7uttg+EgyYxspypRcF8AMkgeI4/GjGybrktFz49xvkP6Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ddb62db1403137715ab4e1bd76c38afd39cbe675a9b932cec4298bb55291dc38","last_reissued_at":"2026-07-05T09:36:05.054816Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:36:05.054816Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Llama Guard 3 Vision: Safeguarding Human-AI Image Understanding Conversations","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Eric Smith, Hongyuan Zhan, Javier Rando, Jianfeng Chi, Kartikeya Upasani, Kate Plawiak, Mahesh Pasupuleti, Ujjwal Karn, Yiming Zhang, Zacharie Delpierre Coudert","submitted_at":"2024-11-15T18:34:07Z","abstract_excerpt":"We introduce Llama Guard 3 Vision, a multimodal LLM-based safeguard for human-AI conversations that involves image understanding: it can be used to safeguard content for both multimodal LLM inputs (prompt classification) and outputs (response classification). Unlike the previous text-only Llama Guard versions (Inan et al., 2023; Llama Team, 2024b,a), it is specifically designed to support image reasoning use cases and is optimized to detect harmful multimodal (text and image) prompts and text responses to these prompts. Llama Guard 3 Vision is fine-tuned on Llama 3.2-Vision and demonstrates st"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.10414","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.10414/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.10414","created_at":"2026-07-05T09:36:05.054889+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.10414v1","created_at":"2026-07-05T09:36:05.054889+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.10414","created_at":"2026-07-05T09:36:05.054889+00:00"},{"alias_kind":"pith_short_12","alias_value":"3W3C3MKAGE3X","created_at":"2026-07-05T09:36:05.054889+00:00"},{"alias_kind":"pith_short_16","alias_value":"3W3C3MKAGE3XCWVU","created_at":"2026-07-05T09:36:05.054889+00:00"},{"alias_kind":"pith_short_8","alias_value":"3W3C3MKA","created_at":"2026-07-05T09:36:05.054889+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":20,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2604.18847","citing_title":"Human-Guided Harm Recovery for Computer Use Agents","ref_index":8,"is_internal_anchor":true},{"citing_arxiv_id":"2606.25034","citing_title":"Yuvion VL: A Multimodal Foundation Model for Adversarial Content and AI Safety","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26199","citing_title":"MIRAGE: Protecting against Malicious Image Editing via False Moderation","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09700","citing_title":"What the Eyes See, the LLMs Miss: Exploiting Human Perception for Adversarial Text Attacks","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01837","citing_title":"Benign Inputs, Harmful Outputs: Cross-Modal Jailbreaking via Distributed Semantic Recomposition","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02041","citing_title":"SentGuard: Sentence-Level Streaming Guardrails for Large Language Models","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28153","citing_title":"Robust Harmful Features Under Jailbreak Attacks: Mechanistic Evidence from Attention Head Specialization in Large Language Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15030","citing_title":"WARD: Adversarially Robust Defense of Web Agents Against Prompt Injections","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26199","citing_title":"MIRAGE: Protecting against Malicious Image Editing via False Moderation","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25034","citing_title":"Yuvion VL: A Multimodal Foundation Model for Adversarial Content and AI Safety","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2502.01241","citing_title":"Peering Behind the Shield: Guardrail Identification in Large Language Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2503.03480","citing_title":"SafeVLA: Towards Safety Alignment of Vision-Language-Action Model via Constrained Learning","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17610","citing_title":"SafeLens: Deliberate and Efficient Video Guardrails with Fast-and-Slow Screening","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18915","citing_title":"DMN: A Compositional Framework for Jailbreaking Multimodal LLMs with Multi-Image Inputs","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2512.12069","citing_title":"Rethinking Jailbreak Detection of Large Vision Language Models with Representational Contrastive Scoring","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2603.27112","citing_title":"RailVQA: A Benchmark and Framework for Efficient Interpretable Visual Cognition in Automatic Train Operation","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04785","citing_title":"AgentTrust: Runtime Safety Evaluation and Interception for AI Agent Tool Use","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18847","citing_title":"Human-Guided Harm Recovery for Computer Use Agents","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08846","citing_title":"Dictionary-Aligned Concept Control for Safeguarding Multimodal LLMs","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17053","citing_title":"Jailbreaking Large Language Models with Morality Attacks","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3W3C3MKAGE3XCWVU4G6XNQ4K7U","json":"https://pith.science/pith/3W3C3MKAGE3XCWVU4G6XNQ4K7U.json","graph_json":"https://pith.science/api/pith-number/3W3C3MKAGE3XCWVU4G6XNQ4K7U/graph.json","events_json":"https://pith.science/api/pith-number/3W3C3MKAGE3XCWVU4G6XNQ4K7U/events.json","paper":"https://pith.science/paper/3W3C3MKA"},"agent_actions":{"view_html":"https://pith.science/pith/3W3C3MKAGE3XCWVU4G6XNQ4K7U","download_json":"https://pith.science/pith/3W3C3MKAGE3XCWVU4G6XNQ4K7U.json","view_paper":"https://pith.science/paper/3W3C3MKA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.10414&json=true","fetch_graph":"https://pith.science/api/pith-number/3W3C3MKAGE3XCWVU4G6XNQ4K7U/graph.json","fetch_events":"https://pith.science/api/pith-number/3W3C3MKAGE3XCWVU4G6XNQ4K7U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3W3C3MKAGE3XCWVU4G6XNQ4K7U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3W3C3MKAGE3XCWVU4G6XNQ4K7U/action/storage_attestation","attest_author":"https://pith.science/pith/3W3C3MKAGE3XCWVU4G6XNQ4K7U/action/author_attestation","sign_citation":"https://pith.science/pith/3W3C3MKAGE3XCWVU4G6XNQ4K7U/action/citation_signature","submit_replication":"https://pith.science/pith/3W3C3MKAGE3XCWVU4G6XNQ4K7U/action/replication_record"}},"created_at":"2026-07-05T09:36:05.054889+00:00","updated_at":"2026-07-05T09:36:05.054889+00:00"}