{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:WILDKNTLPFWDBSQFDHKUU463W3","short_pith_number":"pith:WILDKNTL","schema_version":"1.0","canonical_sha256":"b21635366b796c30ca0519d54a73dbb6c87b91a4c3050e586867199fe49a4c38","source":{"kind":"arxiv","id":"2411.17713","version":1},"attestation_state":"computed","paper":{"title":"Llama Guard 3-1B-INT4: Compact and Efficient Safeguard for Human-AI Conversations","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.DC","authors_text":"Bilge Soran, Changsheng Zhao, Eric Smith, Hongyuan Zhan, Igor Fedorov, Jianfeng Chi, Kate Plawiak, Kimish Patel, Lemeng Wu, Mahesh Pasupuleti, Naveen Suda, Rachad Alao, Raghuraman Krishnamoorthi, Tarek Elgamal, Tijmen Blankevoort, Vikas Chandra, Yangyang Shi, Yuriy Hulovatyy, Zacharie Delpierre Coudert, Zechun Liu","submitted_at":"2024-11-18T21:42:17Z","abstract_excerpt":"This paper presents Llama Guard 3-1B-INT4, a compact and efficient Llama Guard model, which has been open-sourced to the community during Meta Connect 2024. We demonstrate that Llama Guard 3-1B-INT4 can be deployed on resource-constrained devices, achieving a throughput of at least 30 tokens per second and a time-to-first-token of 2.5 seconds or less on a commodity Android mobile CPU. Notably, our experiments show that Llama Guard 3-1B-INT4 attains comparable or superior safety moderation scores to its larger counterpart, Llama Guard 3-1B, despite being approximately 7 times smaller in size (4"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.17713","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DC","submitted_at":"2024-11-18T21:42:17Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"0720b054a4ddc583444031e47d596ee885dbf10485d53454b822160c4ec4878f","abstract_canon_sha256":"3eb153489b1814874810897f3d6e5fc97ec9514fa28767a0b1d4863bcfa31335"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:41:05.574189Z","signature_b64":"cSRoKKKG1dgL4SogOugb3+FtGPSu+G5VNW+u5YWTwsyEyEyLDA8Vb9eL7moyENPgXBfbTOpg6/bm7iPjnt+tCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b21635366b796c30ca0519d54a73dbb6c87b91a4c3050e586867199fe49a4c38","last_reissued_at":"2026-07-05T09:41:05.573697Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:41:05.573697Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Llama Guard 3-1B-INT4: Compact and Efficient Safeguard for Human-AI Conversations","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.DC","authors_text":"Bilge Soran, Changsheng Zhao, Eric Smith, Hongyuan Zhan, Igor Fedorov, Jianfeng Chi, Kate Plawiak, Kimish Patel, Lemeng Wu, Mahesh Pasupuleti, Naveen Suda, Rachad Alao, Raghuraman Krishnamoorthi, Tarek Elgamal, Tijmen Blankevoort, Vikas Chandra, Yangyang Shi, Yuriy Hulovatyy, Zacharie Delpierre Coudert, Zechun Liu","submitted_at":"2024-11-18T21:42:17Z","abstract_excerpt":"This paper presents Llama Guard 3-1B-INT4, a compact and efficient Llama Guard model, which has been open-sourced to the community during Meta Connect 2024. We demonstrate that Llama Guard 3-1B-INT4 can be deployed on resource-constrained devices, achieving a throughput of at least 30 tokens per second and a time-to-first-token of 2.5 seconds or less on a commodity Android mobile CPU. Notably, our experiments show that Llama Guard 3-1B-INT4 attains comparable or superior safety moderation scores to its larger counterpart, Llama Guard 3-1B, despite being approximately 7 times smaller in size (4"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.17713","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.17713/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.17713","created_at":"2026-07-05T09:41:05.573759+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.17713v1","created_at":"2026-07-05T09:41:05.573759+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.17713","created_at":"2026-07-05T09:41:05.573759+00:00"},{"alias_kind":"pith_short_12","alias_value":"WILDKNTLPFWD","created_at":"2026-07-05T09:41:05.573759+00:00"},{"alias_kind":"pith_short_16","alias_value":"WILDKNTLPFWDBSQF","created_at":"2026-07-05T09:41:05.573759+00:00"},{"alias_kind":"pith_short_8","alias_value":"WILDKNTL","created_at":"2026-07-05T09:41:05.573759+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26377","citing_title":"Verifying Intent and Harm: A Unified Defense Against LLM-Generated Threats","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09388","citing_title":"Distilling Safe LLM Systems via Soft Prompts for On Device Settings","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07968","citing_title":"RecurGuard: Runtime Monitoring for Reasoning-Token Consumption Attacks","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01513","citing_title":"Compliance-Scored Best-of-N Guardrail Orchestration for Multimodal Document Generation in Payments Dispute Defense","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2511.12710","citing_title":"Evolve the Method, Not the Prompts: Evolutionary Synthesis of Jailbreak Attacks on LLMs","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17329","citing_title":"LPG: Balancing Efficiency and Policy Reasoning in Latent Policy Guardrails","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2510.09689","citing_title":"When Search Goes Wrong: Red-Teaming Web-Augmented Large Language Models","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2603.15954","citing_title":"MobileLLM-Flash: Latency-Guided On-Device LLM Design for Industry Scale Deployment","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06552","citing_title":"To Lie or Not to Lie? Investigating The Biased Spread of Global Lies by LLMs","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WILDKNTLPFWDBSQFDHKUU463W3","json":"https://pith.science/pith/WILDKNTLPFWDBSQFDHKUU463W3.json","graph_json":"https://pith.science/api/pith-number/WILDKNTLPFWDBSQFDHKUU463W3/graph.json","events_json":"https://pith.science/api/pith-number/WILDKNTLPFWDBSQFDHKUU463W3/events.json","paper":"https://pith.science/paper/WILDKNTL"},"agent_actions":{"view_html":"https://pith.science/pith/WILDKNTLPFWDBSQFDHKUU463W3","download_json":"https://pith.science/pith/WILDKNTLPFWDBSQFDHKUU463W3.json","view_paper":"https://pith.science/paper/WILDKNTL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.17713&json=true","fetch_graph":"https://pith.science/api/pith-number/WILDKNTLPFWDBSQFDHKUU463W3/graph.json","fetch_events":"https://pith.science/api/pith-number/WILDKNTLPFWDBSQFDHKUU463W3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WILDKNTLPFWDBSQFDHKUU463W3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WILDKNTLPFWDBSQFDHKUU463W3/action/storage_attestation","attest_author":"https://pith.science/pith/WILDKNTLPFWDBSQFDHKUU463W3/action/author_attestation","sign_citation":"https://pith.science/pith/WILDKNTLPFWDBSQFDHKUU463W3/action/citation_signature","submit_replication":"https://pith.science/pith/WILDKNTLPFWDBSQFDHKUU463W3/action/replication_record"}},"created_at":"2026-07-05T09:41:05.573759+00:00","updated_at":"2026-07-05T09:41:05.573759+00:00"}