{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GX4NI6W72P65FQJJIGXG6JQXNH","short_pith_number":"pith:GX4NI6W7","schema_version":"1.0","canonical_sha256":"35f8d47adfd3fdd2c12941ae6f261769d5f339c2a81be57bd702df88703799b8","source":{"kind":"arxiv","id":"2407.06866","version":3},"attestation_state":"computed","paper":{"title":"ChatGPT Doesn't Trust Chargers Fans: Guardrail Sensitivity in Context","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Naomi Saphra, Victoria R. Li, Yida Chen","submitted_at":"2024-07-09T13:53:38Z","abstract_excerpt":"While the biases of language models in production are extensively documented, the biases of their guardrails have been neglected. This paper studies how contextual information about the user influences the likelihood of an LLM to refuse to execute a request. By generating user biographies that offer ideological and demographic information, we find a number of biases in guardrail sensitivity on GPT-3.5. Younger, female, and Asian-American personas are more likely to trigger a refusal guardrail when requesting censored or illegal information. Guardrails are also sycophantic, refusing to comply w"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.06866","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-09T13:53:38Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"7bffb994498938f2e552ed6dcc5fca7e8349fd474520331e913579170ad40932","abstract_canon_sha256":"6d925abf61cce1ebd523703c3443dcf2960ed367edf17e463a6698a00e129ef9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:59:06.090916Z","signature_b64":"4WYWuJw7buk24WQEs34fZjGl2yvrov/nmchWbaRnQpVJXjMJYZ0medt3kUptWLPyO6expOzEohOTeADtdn0LDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"35f8d47adfd3fdd2c12941ae6f261769d5f339c2a81be57bd702df88703799b8","last_reissued_at":"2026-07-05T11:59:06.090444Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:59:06.090444Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ChatGPT Doesn't Trust Chargers Fans: Guardrail Sensitivity in Context","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Naomi Saphra, Victoria R. Li, Yida Chen","submitted_at":"2024-07-09T13:53:38Z","abstract_excerpt":"While the biases of language models in production are extensively documented, the biases of their guardrails have been neglected. This paper studies how contextual information about the user influences the likelihood of an LLM to refuse to execute a request. By generating user biographies that offer ideological and demographic information, we find a number of biases in guardrail sensitivity on GPT-3.5. Younger, female, and Asian-American personas are more likely to trigger a refusal guardrail when requesting censored or illegal information. Guardrails are also sycophantic, refusing to comply w"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.06866","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.06866/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.06866","created_at":"2026-07-05T11:59:06.090502+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.06866v3","created_at":"2026-07-05T11:59:06.090502+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.06866","created_at":"2026-07-05T11:59:06.090502+00:00"},{"alias_kind":"pith_short_12","alias_value":"GX4NI6W72P65","created_at":"2026-07-05T11:59:06.090502+00:00"},{"alias_kind":"pith_short_16","alias_value":"GX4NI6W72P65FQJJ","created_at":"2026-07-05T11:59:06.090502+00:00"},{"alias_kind":"pith_short_8","alias_value":"GX4NI6W7","created_at":"2026-07-05T11:59:06.090502+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.30085","citing_title":"Not-quite-human tastes: the stylized omnivorousness of LLM survey surrogates","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2504.04927","citing_title":"Creating and Evaluating Personas Using Generative AI: A Scoping Review of 81 Articles","ref_index":62,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GX4NI6W72P65FQJJIGXG6JQXNH","json":"https://pith.science/pith/GX4NI6W72P65FQJJIGXG6JQXNH.json","graph_json":"https://pith.science/api/pith-number/GX4NI6W72P65FQJJIGXG6JQXNH/graph.json","events_json":"https://pith.science/api/pith-number/GX4NI6W72P65FQJJIGXG6JQXNH/events.json","paper":"https://pith.science/paper/GX4NI6W7"},"agent_actions":{"view_html":"https://pith.science/pith/GX4NI6W72P65FQJJIGXG6JQXNH","download_json":"https://pith.science/pith/GX4NI6W72P65FQJJIGXG6JQXNH.json","view_paper":"https://pith.science/paper/GX4NI6W7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.06866&json=true","fetch_graph":"https://pith.science/api/pith-number/GX4NI6W72P65FQJJIGXG6JQXNH/graph.json","fetch_events":"https://pith.science/api/pith-number/GX4NI6W72P65FQJJIGXG6JQXNH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GX4NI6W72P65FQJJIGXG6JQXNH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GX4NI6W72P65FQJJIGXG6JQXNH/action/storage_attestation","attest_author":"https://pith.science/pith/GX4NI6W72P65FQJJIGXG6JQXNH/action/author_attestation","sign_citation":"https://pith.science/pith/GX4NI6W72P65FQJJIGXG6JQXNH/action/citation_signature","submit_replication":"https://pith.science/pith/GX4NI6W72P65FQJJIGXG6JQXNH/action/replication_record"}},"created_at":"2026-07-05T11:59:06.090502+00:00","updated_at":"2026-07-05T11:59:06.090502+00:00"}