{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:Z2RPNF3HEHMRNELWYE3SHD55DM","short_pith_number":"pith:Z2RPNF3H","schema_version":"1.0","canonical_sha256":"cea2f6976721d9169176c137238fbd1b00c340bbdf40b9d3e0d3415b902aeb1f","source":{"kind":"arxiv","id":"2310.14303","version":2},"attestation_state":"computed","paper":{"title":"Language Model Unalignment: Parametric Red-Teaming to Expose Hidden Harms and Biases","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Rishabh Bhardwaj, Soujanya Poria","submitted_at":"2023-10-22T13:55:46Z","abstract_excerpt":"Red-teaming has been a widely adopted way to evaluate the harmfulness of Large Language Models (LLMs). It aims to jailbreak a model's safety behavior to make it act as a helpful agent disregarding the harmfulness of the query. Existing methods are primarily based on input text-based red-teaming such as adversarial prompts, low-resource prompts, or contextualized prompts to condition the model in a way to bypass its safe behavior. Bypassing the guardrails uncovers hidden harmful information and biases in the model that are left untreated or newly introduced by its safety training. However, prom"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.14303","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2023-10-22T13:55:46Z","cross_cats_sorted":[],"title_canon_sha256":"6094789d24faa047509eed177ea31c0201b7441990b829a20aad6824765d172b","abstract_canon_sha256":"becc16fef51bae47a8a79afb5433ec840ba34860ac94fc9fc1632fff789fb717"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:12:00.556309Z","signature_b64":"ofXZJGe2ws7WNnzXkKafHmS+cn4kYgAzS6c06KQAc2ecnDPtDIaDfYYWISmwTqbMgak8BoEJ20QFN7hHHPxLDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cea2f6976721d9169176c137238fbd1b00c340bbdf40b9d3e0d3415b902aeb1f","last_reissued_at":"2026-07-05T07:12:00.555881Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:12:00.555881Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Language Model Unalignment: Parametric Red-Teaming to Expose Hidden Harms and Biases","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Rishabh Bhardwaj, Soujanya Poria","submitted_at":"2023-10-22T13:55:46Z","abstract_excerpt":"Red-teaming has been a widely adopted way to evaluate the harmfulness of Large Language Models (LLMs). It aims to jailbreak a model's safety behavior to make it act as a helpful agent disregarding the harmfulness of the query. Existing methods are primarily based on input text-based red-teaming such as adversarial prompts, low-resource prompts, or contextualized prompts to condition the model in a way to bypass its safe behavior. Bypassing the guardrails uncovers hidden harmful information and biases in the model that are left untreated or newly introduced by its safety training. However, prom"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.14303","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.14303/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.14303","created_at":"2026-07-05T07:12:00.555936+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.14303v2","created_at":"2026-07-05T07:12:00.555936+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.14303","created_at":"2026-07-05T07:12:00.555936+00:00"},{"alias_kind":"pith_short_12","alias_value":"Z2RPNF3HEHMR","created_at":"2026-07-05T07:12:00.555936+00:00"},{"alias_kind":"pith_short_16","alias_value":"Z2RPNF3HEHMRNELW","created_at":"2026-07-05T07:12:00.555936+00:00"},{"alias_kind":"pith_short_8","alias_value":"Z2RPNF3H","created_at":"2026-07-05T07:12:00.555936+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.24902","citing_title":"Safety Drift After Fine-Tuning: Evidence from High-Stakes Domains","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Z2RPNF3HEHMRNELWYE3SHD55DM","json":"https://pith.science/pith/Z2RPNF3HEHMRNELWYE3SHD55DM.json","graph_json":"https://pith.science/api/pith-number/Z2RPNF3HEHMRNELWYE3SHD55DM/graph.json","events_json":"https://pith.science/api/pith-number/Z2RPNF3HEHMRNELWYE3SHD55DM/events.json","paper":"https://pith.science/paper/Z2RPNF3H"},"agent_actions":{"view_html":"https://pith.science/pith/Z2RPNF3HEHMRNELWYE3SHD55DM","download_json":"https://pith.science/pith/Z2RPNF3HEHMRNELWYE3SHD55DM.json","view_paper":"https://pith.science/paper/Z2RPNF3H","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.14303&json=true","fetch_graph":"https://pith.science/api/pith-number/Z2RPNF3HEHMRNELWYE3SHD55DM/graph.json","fetch_events":"https://pith.science/api/pith-number/Z2RPNF3HEHMRNELWYE3SHD55DM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Z2RPNF3HEHMRNELWYE3SHD55DM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Z2RPNF3HEHMRNELWYE3SHD55DM/action/storage_attestation","attest_author":"https://pith.science/pith/Z2RPNF3HEHMRNELWYE3SHD55DM/action/author_attestation","sign_citation":"https://pith.science/pith/Z2RPNF3HEHMRNELWYE3SHD55DM/action/citation_signature","submit_replication":"https://pith.science/pith/Z2RPNF3HEHMRNELWYE3SHD55DM/action/replication_record"}},"created_at":"2026-07-05T07:12:00.555936+00:00","updated_at":"2026-07-05T07:12:00.555936+00:00"}