{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:H7PBCME3GDWCRE4ZT4D5NV7GGF","short_pith_number":"pith:H7PBCME3","schema_version":"1.0","canonical_sha256":"3fde11309b30ec2893999f07d6d7e63167ec43b77c0e14bddd56feb87417c657","source":{"kind":"arxiv","id":"2410.02684","version":1},"attestation_state":"computed","paper":{"title":"HiddenGuard: Fine-Grained Safe Generation with Specialized Representation Router","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Baolong Bi, Lingrui Mei, Ruibin Yuan, Shenghua Liu, Xueqi Cheng, Yiwei Wang","submitted_at":"2024-10-03T17:10:41Z","abstract_excerpt":"As Large Language Models (LLMs) grow increasingly powerful, ensuring their safety and alignment with human values remains a critical challenge. Ideally, LLMs should provide informative responses while avoiding the disclosure of harmful or sensitive information. However, current alignment approaches, which rely heavily on refusal strategies, such as training models to completely reject harmful prompts or applying coarse filters are limited by their binary nature. These methods either fully deny access to information or grant it without sufficient nuance, leading to overly cautious responses or "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.02684","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-03T17:10:41Z","cross_cats_sorted":[],"title_canon_sha256":"843508566bc463db5598aed43c18df45cd3e722f9f5441734c6543480b3f277b","abstract_canon_sha256":"e8f5b6d835db04dd61f9d95bb8083fca9fb2e4058e374da69f58bcb778bb2a89"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:15:24.204191Z","signature_b64":"Q38xXYR+cx5q0nk3b6+niro4DlP46SWtZwz9C1t87RG7X8l9opQUSPF/dOING61pkL9XvIGTBS4hB94BD7esCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3fde11309b30ec2893999f07d6d7e63167ec43b77c0e14bddd56feb87417c657","last_reissued_at":"2026-07-05T09:15:24.203705Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:15:24.203705Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HiddenGuard: Fine-Grained Safe Generation with Specialized Representation Router","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Baolong Bi, Lingrui Mei, Ruibin Yuan, Shenghua Liu, Xueqi Cheng, Yiwei Wang","submitted_at":"2024-10-03T17:10:41Z","abstract_excerpt":"As Large Language Models (LLMs) grow increasingly powerful, ensuring their safety and alignment with human values remains a critical challenge. Ideally, LLMs should provide informative responses while avoiding the disclosure of harmful or sensitive information. However, current alignment approaches, which rely heavily on refusal strategies, such as training models to completely reject harmful prompts or applying coarse filters are limited by their binary nature. These methods either fully deny access to information or grant it without sufficient nuance, leading to overly cautious responses or "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.02684","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.02684/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.02684","created_at":"2026-07-05T09:15:24.203762+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.02684v1","created_at":"2026-07-05T09:15:24.203762+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.02684","created_at":"2026-07-05T09:15:24.203762+00:00"},{"alias_kind":"pith_short_12","alias_value":"H7PBCME3GDWC","created_at":"2026-07-05T09:15:24.203762+00:00"},{"alias_kind":"pith_short_16","alias_value":"H7PBCME3GDWCRE4Z","created_at":"2026-07-05T09:15:24.203762+00:00"},{"alias_kind":"pith_short_8","alias_value":"H7PBCME3","created_at":"2026-07-05T09:15:24.203762+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.23974","citing_title":"AERIC: Anticipatory Hidden-State Monitoring for Implicit Harmful Dialogue","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05418","citing_title":"VideoStir: Understanding Long Videos via Spatio-Temporally Structured and Intent-Aware RAG","ref_index":61,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/H7PBCME3GDWCRE4ZT4D5NV7GGF","json":"https://pith.science/pith/H7PBCME3GDWCRE4ZT4D5NV7GGF.json","graph_json":"https://pith.science/api/pith-number/H7PBCME3GDWCRE4ZT4D5NV7GGF/graph.json","events_json":"https://pith.science/api/pith-number/H7PBCME3GDWCRE4ZT4D5NV7GGF/events.json","paper":"https://pith.science/paper/H7PBCME3"},"agent_actions":{"view_html":"https://pith.science/pith/H7PBCME3GDWCRE4ZT4D5NV7GGF","download_json":"https://pith.science/pith/H7PBCME3GDWCRE4ZT4D5NV7GGF.json","view_paper":"https://pith.science/paper/H7PBCME3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.02684&json=true","fetch_graph":"https://pith.science/api/pith-number/H7PBCME3GDWCRE4ZT4D5NV7GGF/graph.json","fetch_events":"https://pith.science/api/pith-number/H7PBCME3GDWCRE4ZT4D5NV7GGF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/H7PBCME3GDWCRE4ZT4D5NV7GGF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/H7PBCME3GDWCRE4ZT4D5NV7GGF/action/storage_attestation","attest_author":"https://pith.science/pith/H7PBCME3GDWCRE4ZT4D5NV7GGF/action/author_attestation","sign_citation":"https://pith.science/pith/H7PBCME3GDWCRE4ZT4D5NV7GGF/action/citation_signature","submit_replication":"https://pith.science/pith/H7PBCME3GDWCRE4ZT4D5NV7GGF/action/replication_record"}},"created_at":"2026-07-05T09:15:24.203762+00:00","updated_at":"2026-07-05T09:15:24.203762+00:00"}