{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:TLLOITWX3CXXQLX4AWA3HMQHFS","short_pith_number":"pith:TLLOITWX","schema_version":"1.0","canonical_sha256":"9ad6e44ed7d8af782efc0581b3b2072c9b8dd564c970dc347fff19388f731de5","source":{"kind":"arxiv","id":"2505.23848","version":1},"attestation_state":"computed","paper":{"title":"Derailing Non-Answers via Logit Suppression at Output Subspace Boundaries in RLHF-Aligned Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Ganesh Gopalakrishnan, Harvey Dam, Jonas Knochelmann, Vinu Joseph","submitted_at":"2025-05-28T20:25:24Z","abstract_excerpt":"We introduce a method to reduce refusal rates of large language models (LLMs) on sensitive content without modifying model weights or prompts. Motivated by the observation that refusals in certain models were often preceded by the specific token sequence of a token marking the beginning of the chain-of-thought (CoT) block (<think>) followed by a double newline token (\\n\\n), we investigate the impact of two simple formatting adjustments during generation: suppressing \\n\\n after <think> and suppressing the end-of-sequence token after the end of the CoT block (</think>). Our method requires no da"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.23848","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-05-28T20:25:24Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"6a9d9c5dd93fa8c34196ceca5b22fa5b709b7dc490e777bbe6406b6fd513cf5a","abstract_canon_sha256":"d823429fcb20d7d01f0cec7b7bd309a54ae0bd43a8b686e398bb7746627e6051"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:12:36.668121Z","signature_b64":"YnjjNz1tVwYLVw/BcScpPexDuByQ2X9ExZ0hTN3rCGr2doUF6NhoN6GibWiCiLwGb9/rC+wnb6cRpJxqnSewAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9ad6e44ed7d8af782efc0581b3b2072c9b8dd564c970dc347fff19388f731de5","last_reissued_at":"2026-07-05T11:12:36.666771Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:12:36.666771Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Derailing Non-Answers via Logit Suppression at Output Subspace Boundaries in RLHF-Aligned Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Ganesh Gopalakrishnan, Harvey Dam, Jonas Knochelmann, Vinu Joseph","submitted_at":"2025-05-28T20:25:24Z","abstract_excerpt":"We introduce a method to reduce refusal rates of large language models (LLMs) on sensitive content without modifying model weights or prompts. Motivated by the observation that refusals in certain models were often preceded by the specific token sequence of a token marking the beginning of the chain-of-thought (CoT) block (<think>) followed by a double newline token (\\n\\n), we investigate the impact of two simple formatting adjustments during generation: suppressing \\n\\n after <think> and suppressing the end-of-sequence token after the end of the CoT block (</think>). Our method requires no da"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.23848","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.23848/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.23848","created_at":"2026-07-05T11:12:36.667369+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.23848v1","created_at":"2026-07-05T11:12:36.667369+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.23848","created_at":"2026-07-05T11:12:36.667369+00:00"},{"alias_kind":"pith_short_12","alias_value":"TLLOITWX3CXX","created_at":"2026-07-05T11:12:36.667369+00:00"},{"alias_kind":"pith_short_16","alias_value":"TLLOITWX3CXXQLX4","created_at":"2026-07-05T11:12:36.667369+00:00"},{"alias_kind":"pith_short_8","alias_value":"TLLOITWX","created_at":"2026-07-05T11:12:36.667369+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TLLOITWX3CXXQLX4AWA3HMQHFS","json":"https://pith.science/pith/TLLOITWX3CXXQLX4AWA3HMQHFS.json","graph_json":"https://pith.science/api/pith-number/TLLOITWX3CXXQLX4AWA3HMQHFS/graph.json","events_json":"https://pith.science/api/pith-number/TLLOITWX3CXXQLX4AWA3HMQHFS/events.json","paper":"https://pith.science/paper/TLLOITWX"},"agent_actions":{"view_html":"https://pith.science/pith/TLLOITWX3CXXQLX4AWA3HMQHFS","download_json":"https://pith.science/pith/TLLOITWX3CXXQLX4AWA3HMQHFS.json","view_paper":"https://pith.science/paper/TLLOITWX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.23848&json=true","fetch_graph":"https://pith.science/api/pith-number/TLLOITWX3CXXQLX4AWA3HMQHFS/graph.json","fetch_events":"https://pith.science/api/pith-number/TLLOITWX3CXXQLX4AWA3HMQHFS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TLLOITWX3CXXQLX4AWA3HMQHFS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TLLOITWX3CXXQLX4AWA3HMQHFS/action/storage_attestation","attest_author":"https://pith.science/pith/TLLOITWX3CXXQLX4AWA3HMQHFS/action/author_attestation","sign_citation":"https://pith.science/pith/TLLOITWX3CXXQLX4AWA3HMQHFS/action/citation_signature","submit_replication":"https://pith.science/pith/TLLOITWX3CXXQLX4AWA3HMQHFS/action/replication_record"}},"created_at":"2026-07-05T11:12:36.667369+00:00","updated_at":"2026-07-05T11:12:36.667369+00:00"}