{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3DMTQWZENBPXGZNL3DVHQASWSO","short_pith_number":"pith:3DMTQWZE","schema_version":"1.0","canonical_sha256":"d8d9385b24685f7365abd8ea78025693971351a1cebbd51728a047ade433a383","source":{"kind":"arxiv","id":"2412.06748","version":2},"attestation_state":"computed","paper":{"title":"Refusal Tokens: A Simple Way to Calibrate Refusals in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Aditya Shrivastava, Alfy Samuel, Anoop Kumar, Ashwinee Panda, Chenyang Zhu, Daben Liu, Micah Goldblum, Neel Jain, Tom Goldstein","submitted_at":"2024-12-09T18:40:44Z","abstract_excerpt":"A key component of building safe and reliable language models is enabling the models to appropriately refuse to follow certain instructions or answer certain questions. We may want models to output refusal messages for various categories of user queries, for example, ill-posed questions, instructions for committing illegal acts, or queries which require information past the model's knowledge horizon. Engineering models that refuse to answer such questions is complicated by the fact that an individual may want their model to exhibit varying levels of sensitivity for refusing queries of various "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.06748","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-12-09T18:40:44Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"112b48b1d5c2db3046135dff5619765a26294ea51717b000ec7f8051d8fd11d6","abstract_canon_sha256":"3630c25499afc3b0f6be5437326e7df57c7cba3d5455df8398be1c3cfc97b168"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:01:43.905779Z","signature_b64":"+S4vfQArSPh7t/c65blPA11FT/NI0uaF4dGWNxnVL2hEeMZhqp7m5FD7iGXhGG5YHEfb4eTuzO71AJPFFQLoDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d8d9385b24685f7365abd8ea78025693971351a1cebbd51728a047ade433a383","last_reissued_at":"2026-07-05T12:01:43.905300Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:01:43.905300Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Refusal Tokens: A Simple Way to Calibrate Refusals in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Aditya Shrivastava, Alfy Samuel, Anoop Kumar, Ashwinee Panda, Chenyang Zhu, Daben Liu, Micah Goldblum, Neel Jain, Tom Goldstein","submitted_at":"2024-12-09T18:40:44Z","abstract_excerpt":"A key component of building safe and reliable language models is enabling the models to appropriately refuse to follow certain instructions or answer certain questions. We may want models to output refusal messages for various categories of user queries, for example, ill-posed questions, instructions for committing illegal acts, or queries which require information past the model's knowledge horizon. Engineering models that refuse to answer such questions is complicated by the fact that an individual may want their model to exhibit varying levels of sensitivity for refusing queries of various "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.06748","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.06748/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.06748","created_at":"2026-07-05T12:01:43.905387+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.06748v2","created_at":"2026-07-05T12:01:43.905387+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.06748","created_at":"2026-07-05T12:01:43.905387+00:00"},{"alias_kind":"pith_short_12","alias_value":"3DMTQWZENBPX","created_at":"2026-07-05T12:01:43.905387+00:00"},{"alias_kind":"pith_short_16","alias_value":"3DMTQWZENBPXGZNL","created_at":"2026-07-05T12:01:43.905387+00:00"},{"alias_kind":"pith_short_8","alias_value":"3DMTQWZE","created_at":"2026-07-05T12:01:43.905387+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2508.11222","citing_title":"ORFuzz: Fuzzing the \"Other Side\" of LLM Safety -- Testing Over-Refusal","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03216","citing_title":"BAS: A Decision-Theoretic Approach to Evaluating Large Language Model Confidence","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3DMTQWZENBPXGZNL3DVHQASWSO","json":"https://pith.science/pith/3DMTQWZENBPXGZNL3DVHQASWSO.json","graph_json":"https://pith.science/api/pith-number/3DMTQWZENBPXGZNL3DVHQASWSO/graph.json","events_json":"https://pith.science/api/pith-number/3DMTQWZENBPXGZNL3DVHQASWSO/events.json","paper":"https://pith.science/paper/3DMTQWZE"},"agent_actions":{"view_html":"https://pith.science/pith/3DMTQWZENBPXGZNL3DVHQASWSO","download_json":"https://pith.science/pith/3DMTQWZENBPXGZNL3DVHQASWSO.json","view_paper":"https://pith.science/paper/3DMTQWZE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.06748&json=true","fetch_graph":"https://pith.science/api/pith-number/3DMTQWZENBPXGZNL3DVHQASWSO/graph.json","fetch_events":"https://pith.science/api/pith-number/3DMTQWZENBPXGZNL3DVHQASWSO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3DMTQWZENBPXGZNL3DVHQASWSO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3DMTQWZENBPXGZNL3DVHQASWSO/action/storage_attestation","attest_author":"https://pith.science/pith/3DMTQWZENBPXGZNL3DVHQASWSO/action/author_attestation","sign_citation":"https://pith.science/pith/3DMTQWZENBPXGZNL3DVHQASWSO/action/citation_signature","submit_replication":"https://pith.science/pith/3DMTQWZENBPXGZNL3DVHQASWSO/action/replication_record"}},"created_at":"2026-07-05T12:01:43.905387+00:00","updated_at":"2026-07-05T12:01:43.905387+00:00"}