{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:W2LF25A6FMWP3PVT7J7EKD5FE5","short_pith_number":"pith:W2LF25A6","schema_version":"1.0","canonical_sha256":"b6965d741e2b2cfdbeb3fa7e450fa52757e052fb466b50a1e81161e4bf3b04b7","source":{"kind":"arxiv","id":"2502.12464","version":5},"attestation_state":"computed","paper":{"title":"SafeRoute: Adaptive Model Selection for Efficient and Accurate Safety Guardrails in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dominik Wagner, Dong Bok Lee, Haebin Seong, Juho Lee, Minki Kang, Seanie Lee, Sung Ju Hwang, Tobias Bocklet","submitted_at":"2025-02-18T02:51:17Z","abstract_excerpt":"Deploying large language models (LLMs) in real-world applications requires robust safety guard models to detect and block harmful user prompts. While large safety guard models achieve strong performance, their computational cost is substantial. To mitigate this, smaller distilled models are used, but they often underperform on \"hard\" examples where the larger model provides accurate predictions. We observe that many inputs can be reliably handled by the smaller model, while only a small fraction require the larger model's capacity. Motivated by this, we propose SafeRoute, a binary router that "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.12464","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-18T02:51:17Z","cross_cats_sorted":[],"title_canon_sha256":"a67c49b497b65f3f9708d0e2695fb79df9d99c7af10689094d5331845059ce3e","abstract_canon_sha256":"ba3386f04a3930d3e45223e62724a3f7056c89154dc0edd64b7106078168e348"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:07:20.606121Z","signature_b64":"pFg8TdOWeFZ7YpwFqzW6Y18Ftp6x0en7l0rJp1Wx6LvPckbkzr3VSKZ5zmb0tEoOG3PY55yYCb0UkfKxnS5GDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b6965d741e2b2cfdbeb3fa7e450fa52757e052fb466b50a1e81161e4bf3b04b7","last_reissued_at":"2026-07-05T11:07:20.605660Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:07:20.605660Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SafeRoute: Adaptive Model Selection for Efficient and Accurate Safety Guardrails in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dominik Wagner, Dong Bok Lee, Haebin Seong, Juho Lee, Minki Kang, Seanie Lee, Sung Ju Hwang, Tobias Bocklet","submitted_at":"2025-02-18T02:51:17Z","abstract_excerpt":"Deploying large language models (LLMs) in real-world applications requires robust safety guard models to detect and block harmful user prompts. While large safety guard models achieve strong performance, their computational cost is substantial. To mitigate this, smaller distilled models are used, but they often underperform on \"hard\" examples where the larger model provides accurate predictions. We observe that many inputs can be reliably handled by the smaller model, while only a small fraction require the larger model's capacity. Motivated by this, we propose SafeRoute, a binary router that "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.12464","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.12464/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.12464","created_at":"2026-07-05T11:07:20.605712+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.12464v5","created_at":"2026-07-05T11:07:20.605712+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.12464","created_at":"2026-07-05T11:07:20.605712+00:00"},{"alias_kind":"pith_short_12","alias_value":"W2LF25A6FMWP","created_at":"2026-07-05T11:07:20.605712+00:00"},{"alias_kind":"pith_short_16","alias_value":"W2LF25A6FMWP3PVT","created_at":"2026-07-05T11:07:20.605712+00:00"},{"alias_kind":"pith_short_8","alias_value":"W2LF25A6","created_at":"2026-07-05T11:07:20.605712+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/W2LF25A6FMWP3PVT7J7EKD5FE5","json":"https://pith.science/pith/W2LF25A6FMWP3PVT7J7EKD5FE5.json","graph_json":"https://pith.science/api/pith-number/W2LF25A6FMWP3PVT7J7EKD5FE5/graph.json","events_json":"https://pith.science/api/pith-number/W2LF25A6FMWP3PVT7J7EKD5FE5/events.json","paper":"https://pith.science/paper/W2LF25A6"},"agent_actions":{"view_html":"https://pith.science/pith/W2LF25A6FMWP3PVT7J7EKD5FE5","download_json":"https://pith.science/pith/W2LF25A6FMWP3PVT7J7EKD5FE5.json","view_paper":"https://pith.science/paper/W2LF25A6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.12464&json=true","fetch_graph":"https://pith.science/api/pith-number/W2LF25A6FMWP3PVT7J7EKD5FE5/graph.json","fetch_events":"https://pith.science/api/pith-number/W2LF25A6FMWP3PVT7J7EKD5FE5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/W2LF25A6FMWP3PVT7J7EKD5FE5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/W2LF25A6FMWP3PVT7J7EKD5FE5/action/storage_attestation","attest_author":"https://pith.science/pith/W2LF25A6FMWP3PVT7J7EKD5FE5/action/author_attestation","sign_citation":"https://pith.science/pith/W2LF25A6FMWP3PVT7J7EKD5FE5/action/citation_signature","submit_replication":"https://pith.science/pith/W2LF25A6FMWP3PVT7J7EKD5FE5/action/replication_record"}},"created_at":"2026-07-05T11:07:20.605712+00:00","updated_at":"2026-07-05T11:07:20.605712+00:00"}