{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QDCTYHPLX6A5DNUKUMSZSKMGGQ","short_pith_number":"pith:QDCTYHPL","schema_version":"1.0","canonical_sha256":"80c53c1debbf81d1b68aa3259929863426e6554d499b2d52964a76c3c223a52d","source":{"kind":"arxiv","id":"2412.07724","version":2},"attestation_state":"computed","paper":{"title":"Granite Guardian","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ambrish Rawat, Elizabeth M. Daly, Erik Miehling, Giandomenico Cornacchia, Giulio Zizzo, Inge Vejsbjerg, Inkit Padhi, Keerthiram Murugesan, Kieran Fraser, Kush R. Varshney, Manish Nagireddy, Mark Purcell, Mart\\'in Santill\\'an Cooper, Michael Desmond, Michael Hind, Muhammad Zaid Hameed, Pierre Dognin, Prasanna Sattigeri, Qian Pan, Subhajit Chaudhury, Tejaswini Pedapati, Werner Geyer, Zahra Ashktorab","submitted_at":"2024-12-10T18:17:02Z","abstract_excerpt":"We introduce the Granite Guardian models, a suite of safeguards designed to provide risk detection for prompts and responses, enabling safe and responsible use in combination with any large language model (LLM). These models offer comprehensive coverage across multiple risk dimensions, including social bias, profanity, violence, sexual content, unethical behavior, jailbreaking, and hallucination-related risks such as context relevance, groundedness, and answer relevance for retrieval-augmented generation (RAG). Trained on a unique dataset combining human annotations from diverse sources and sy"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.07724","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-12-10T18:17:02Z","cross_cats_sorted":[],"title_canon_sha256":"cc3cb56031d2d7734ffca95145884421f2df3853d985483e7d71a9912a018f2a","abstract_canon_sha256":"8e3b9ba2d72e2ccffe7ee834e501f2f4305e10f68053451ad418d6fa0ef417f3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:50:08.191089Z","signature_b64":"DoeDN0UL5cT9WfpAc7GEJpudNWPd6o6yfKZUqNBIZmqJWTofnwlNEaAf/Wc3KmOoGTAGkFS27+WnB5TSH3niCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"80c53c1debbf81d1b68aa3259929863426e6554d499b2d52964a76c3c223a52d","last_reissued_at":"2026-07-05T09:50:08.190582Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:50:08.190582Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Granite Guardian","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ambrish Rawat, Elizabeth M. Daly, Erik Miehling, Giandomenico Cornacchia, Giulio Zizzo, Inge Vejsbjerg, Inkit Padhi, Keerthiram Murugesan, Kieran Fraser, Kush R. Varshney, Manish Nagireddy, Mark Purcell, Mart\\'in Santill\\'an Cooper, Michael Desmond, Michael Hind, Muhammad Zaid Hameed, Pierre Dognin, Prasanna Sattigeri, Qian Pan, Subhajit Chaudhury, Tejaswini Pedapati, Werner Geyer, Zahra Ashktorab","submitted_at":"2024-12-10T18:17:02Z","abstract_excerpt":"We introduce the Granite Guardian models, a suite of safeguards designed to provide risk detection for prompts and responses, enabling safe and responsible use in combination with any large language model (LLM). These models offer comprehensive coverage across multiple risk dimensions, including social bias, profanity, violence, sexual content, unethical behavior, jailbreaking, and hallucination-related risks such as context relevance, groundedness, and answer relevance for retrieval-augmented generation (RAG). Trained on a unique dataset combining human annotations from diverse sources and sy"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.07724","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.07724/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.07724","created_at":"2026-07-05T09:50:08.190660+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.07724v2","created_at":"2026-07-05T09:50:08.190660+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.07724","created_at":"2026-07-05T09:50:08.190660+00:00"},{"alias_kind":"pith_short_12","alias_value":"QDCTYHPLX6A5","created_at":"2026-07-05T09:50:08.190660+00:00"},{"alias_kind":"pith_short_16","alias_value":"QDCTYHPLX6A5DNUK","created_at":"2026-07-05T09:50:08.190660+00:00"},{"alias_kind":"pith_short_8","alias_value":"QDCTYHPL","created_at":"2026-07-05T09:50:08.190660+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01277","citing_title":"Cognitive Firewall: A Proactive, Zero-Trust, Multi-Gate Framework for LLM Safety","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20668","citing_title":"BELLS-O: Evaluating the Operational Trade-offs of LLM Supervision Systems","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09388","citing_title":"Distilling Safe LLM Systems via Soft Prompts for On Device Settings","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03648","citing_title":"Safety Measurements for Fine-tuned LLMs Should be Grounded in Capability","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29659","citing_title":"Opir: Efficient Multi-Task Safety Classification for Toxicity, Jailbreaks, Hate Speech, and Harmful Content","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2506.00166","citing_title":"Disentangled Safety Adapters Enable Efficient Guardrails and Flexible Inference-Time Alignment","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2602.07954","citing_title":"Bielik Guard: Efficient Polish Language Safety Classifiers for LLM Content Moderation","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25109","citing_title":"Structured Security Auditing and Robustness Enhancement for Untrusted Agent Skills","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07655","citing_title":"Guardian-as-an-Advisor: Advancing Next-Generation Guardian Models for Trustworthy LLMs","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02914","citing_title":"When Safety Geometry Collapses: Fine-Tuning Vulnerabilities in Agentic Guard Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16542","citing_title":"TWGuard: A Case Study of LLM Safety Guardrails for Localized Linguistic Contexts","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15945","citing_title":"RAGognizer: Hallucination-Aware Fine-Tuning via Detection Head Integration","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QDCTYHPLX6A5DNUKUMSZSKMGGQ","json":"https://pith.science/pith/QDCTYHPLX6A5DNUKUMSZSKMGGQ.json","graph_json":"https://pith.science/api/pith-number/QDCTYHPLX6A5DNUKUMSZSKMGGQ/graph.json","events_json":"https://pith.science/api/pith-number/QDCTYHPLX6A5DNUKUMSZSKMGGQ/events.json","paper":"https://pith.science/paper/QDCTYHPL"},"agent_actions":{"view_html":"https://pith.science/pith/QDCTYHPLX6A5DNUKUMSZSKMGGQ","download_json":"https://pith.science/pith/QDCTYHPLX6A5DNUKUMSZSKMGGQ.json","view_paper":"https://pith.science/paper/QDCTYHPL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.07724&json=true","fetch_graph":"https://pith.science/api/pith-number/QDCTYHPLX6A5DNUKUMSZSKMGGQ/graph.json","fetch_events":"https://pith.science/api/pith-number/QDCTYHPLX6A5DNUKUMSZSKMGGQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QDCTYHPLX6A5DNUKUMSZSKMGGQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QDCTYHPLX6A5DNUKUMSZSKMGGQ/action/storage_attestation","attest_author":"https://pith.science/pith/QDCTYHPLX6A5DNUKUMSZSKMGGQ/action/author_attestation","sign_citation":"https://pith.science/pith/QDCTYHPLX6A5DNUKUMSZSKMGGQ/action/citation_signature","submit_replication":"https://pith.science/pith/QDCTYHPLX6A5DNUKUMSZSKMGGQ/action/replication_record"}},"created_at":"2026-07-05T09:50:08.190660+00:00","updated_at":"2026-07-05T09:50:08.190660+00:00"}