{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GZDJXQIDYOSYJCF53I5GTOOQ7G","short_pith_number":"pith:GZDJXQID","schema_version":"1.0","canonical_sha256":"36469bc103c3a58488bdda3a69b9d0f9ad911a36f85154b2fc9bac320abf2f67","source":{"kind":"arxiv","id":"2410.10414","version":2},"attestation_state":"computed","paper":{"title":"On Calibration of LLM-based Guard Models for Reliable Content Moderation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CR","authors_text":"Hao Wang, Hengguan Huang, Hongfu Liu, Xiangming Gu, Ye Wang","submitted_at":"2024-10-14T12:04:06Z","abstract_excerpt":"Large language models (LLMs) pose significant risks due to the potential for generating harmful content or users attempting to evade guardrails. Existing studies have developed LLM-based guard models designed to moderate the input and output of threat LLMs, ensuring adherence to safety policies by blocking content that violates these protocols upon deployment. However, limited attention has been given to the reliability and calibration of such guard models. In this work, we empirically conduct comprehensive investigations of confidence calibration for 9 existing LLM-based guard models on 12 be"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.10414","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2024-10-14T12:04:06Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"8728b7ec8d40dfa83c619dfcff2ab149a3583f26f3a4401ff40bebbdc3aa8051","abstract_canon_sha256":"03889e721b401c40fd5e53fa252a64bccd46c74317c197f9d4fe8e91a8bd4e4d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:18:43.266832Z","signature_b64":"Xb3mlu/Fro7MwsR9NWNS6Gy1qKkqNNbFrdeUnjBkkvIo2imwvyHse3VDt94w1hSFCvig4ekQaHWPfisi2sVeCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"36469bc103c3a58488bdda3a69b9d0f9ad911a36f85154b2fc9bac320abf2f67","last_reissued_at":"2026-07-05T10:18:43.266275Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:18:43.266275Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On Calibration of LLM-based Guard Models for Reliable Content Moderation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CR","authors_text":"Hao Wang, Hengguan Huang, Hongfu Liu, Xiangming Gu, Ye Wang","submitted_at":"2024-10-14T12:04:06Z","abstract_excerpt":"Large language models (LLMs) pose significant risks due to the potential for generating harmful content or users attempting to evade guardrails. Existing studies have developed LLM-based guard models designed to moderate the input and output of threat LLMs, ensuring adherence to safety policies by blocking content that violates these protocols upon deployment. However, limited attention has been given to the reliability and calibration of such guard models. In this work, we empirically conduct comprehensive investigations of confidence calibration for 9 existing LLM-based guard models on 12 be"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.10414","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.10414/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.10414","created_at":"2026-07-05T10:18:43.266354+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.10414v2","created_at":"2026-07-05T10:18:43.266354+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.10414","created_at":"2026-07-05T10:18:43.266354+00:00"},{"alias_kind":"pith_short_12","alias_value":"GZDJXQIDYOSY","created_at":"2026-07-05T10:18:43.266354+00:00"},{"alias_kind":"pith_short_16","alias_value":"GZDJXQIDYOSYJCF5","created_at":"2026-07-05T10:18:43.266354+00:00"},{"alias_kind":"pith_short_8","alias_value":"GZDJXQID","created_at":"2026-07-05T10:18:43.266354+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22659","citing_title":"Confidently Wrong: Severity-Aware Calibration of Prompt-Injection Detectors under Attack Shift","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GZDJXQIDYOSYJCF53I5GTOOQ7G","json":"https://pith.science/pith/GZDJXQIDYOSYJCF53I5GTOOQ7G.json","graph_json":"https://pith.science/api/pith-number/GZDJXQIDYOSYJCF53I5GTOOQ7G/graph.json","events_json":"https://pith.science/api/pith-number/GZDJXQIDYOSYJCF53I5GTOOQ7G/events.json","paper":"https://pith.science/paper/GZDJXQID"},"agent_actions":{"view_html":"https://pith.science/pith/GZDJXQIDYOSYJCF53I5GTOOQ7G","download_json":"https://pith.science/pith/GZDJXQIDYOSYJCF53I5GTOOQ7G.json","view_paper":"https://pith.science/paper/GZDJXQID","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.10414&json=true","fetch_graph":"https://pith.science/api/pith-number/GZDJXQIDYOSYJCF53I5GTOOQ7G/graph.json","fetch_events":"https://pith.science/api/pith-number/GZDJXQIDYOSYJCF53I5GTOOQ7G/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GZDJXQIDYOSYJCF53I5GTOOQ7G/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GZDJXQIDYOSYJCF53I5GTOOQ7G/action/storage_attestation","attest_author":"https://pith.science/pith/GZDJXQIDYOSYJCF53I5GTOOQ7G/action/author_attestation","sign_citation":"https://pith.science/pith/GZDJXQIDYOSYJCF53I5GTOOQ7G/action/citation_signature","submit_replication":"https://pith.science/pith/GZDJXQIDYOSYJCF53I5GTOOQ7G/action/replication_record"}},"created_at":"2026-07-05T10:18:43.266354+00:00","updated_at":"2026-07-05T10:18:43.266354+00:00"}