{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:E6S44KXPKU2POIFRZIK76TYR7Y","short_pith_number":"pith:E6S44KXP","schema_version":"1.0","canonical_sha256":"27a5ce2aef5534f720b1ca15ff4f11fe3287f0bf2e8c815aff723a83e2407e39","source":{"kind":"arxiv","id":"2306.03423","version":2},"attestation_state":"computed","paper":{"title":"I'm Afraid I Can't Do That: Predicting Prompt Refusal in Black-Box Generative Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Max Reuter, William Schulze","submitted_at":"2023-06-06T05:50:58Z","abstract_excerpt":"Since the release of OpenAI's ChatGPT, generative language models have attracted extensive public attention. The increased usage has highlighted generative models' broad utility, but also revealed several forms of embedded bias. Some is induced by the pre-training corpus; but additional bias specific to generative models arises from the use of subjective fine-tuning to avoid generating harmful content. Fine-tuning bias may come from individual engineers and company policies, and affects which prompts the model chooses to refuse. In this experiment, we characterize ChatGPT's refusal behavior us"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.03423","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2023-06-06T05:50:58Z","cross_cats_sorted":[],"title_canon_sha256":"37d09ad1146a40108190b8dc69d853d69f0383770510eadf1e98d06f443322a9","abstract_canon_sha256":"b51b9cf31962912120647ee526494b89aace2617c49737e3b794efd4aa42ecb7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:20:44.748867Z","signature_b64":"As1t+zlD0fD6yk0+OuXLtpd49WGcaKl+kjZ5fFQiEaqaFRjbyYyaOARKI0BFnGoWeATUnSpoK8Ra0/E69XYmDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"27a5ce2aef5534f720b1ca15ff4f11fe3287f0bf2e8c815aff723a83e2407e39","last_reissued_at":"2026-07-05T06:20:44.748408Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:20:44.748408Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"I'm Afraid I Can't Do That: Predicting Prompt Refusal in Black-Box Generative Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Max Reuter, William Schulze","submitted_at":"2023-06-06T05:50:58Z","abstract_excerpt":"Since the release of OpenAI's ChatGPT, generative language models have attracted extensive public attention. The increased usage has highlighted generative models' broad utility, but also revealed several forms of embedded bias. Some is induced by the pre-training corpus; but additional bias specific to generative models arises from the use of subjective fine-tuning to avoid generating harmful content. Fine-tuning bias may come from individual engineers and company policies, and affects which prompts the model chooses to refuse. In this experiment, we characterize ChatGPT's refusal behavior us"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.03423","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.03423/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.03423","created_at":"2026-07-05T06:20:44.748482+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.03423v2","created_at":"2026-07-05T06:20:44.748482+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.03423","created_at":"2026-07-05T06:20:44.748482+00:00"},{"alias_kind":"pith_short_12","alias_value":"E6S44KXPKU2P","created_at":"2026-07-05T06:20:44.748482+00:00"},{"alias_kind":"pith_short_16","alias_value":"E6S44KXPKU2POIFR","created_at":"2026-07-05T06:20:44.748482+00:00"},{"alias_kind":"pith_short_8","alias_value":"E6S44KXP","created_at":"2026-07-05T06:20:44.748482+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.18322","citing_title":"Is It Bad to Work All the Time? Cross-Cultural Evaluation of Social Norm Biases in GPT-4","ref_index":2025,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/E6S44KXPKU2POIFRZIK76TYR7Y","json":"https://pith.science/pith/E6S44KXPKU2POIFRZIK76TYR7Y.json","graph_json":"https://pith.science/api/pith-number/E6S44KXPKU2POIFRZIK76TYR7Y/graph.json","events_json":"https://pith.science/api/pith-number/E6S44KXPKU2POIFRZIK76TYR7Y/events.json","paper":"https://pith.science/paper/E6S44KXP"},"agent_actions":{"view_html":"https://pith.science/pith/E6S44KXPKU2POIFRZIK76TYR7Y","download_json":"https://pith.science/pith/E6S44KXPKU2POIFRZIK76TYR7Y.json","view_paper":"https://pith.science/paper/E6S44KXP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.03423&json=true","fetch_graph":"https://pith.science/api/pith-number/E6S44KXPKU2POIFRZIK76TYR7Y/graph.json","fetch_events":"https://pith.science/api/pith-number/E6S44KXPKU2POIFRZIK76TYR7Y/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/E6S44KXPKU2POIFRZIK76TYR7Y/action/timestamp_anchor","attest_storage":"https://pith.science/pith/E6S44KXPKU2POIFRZIK76TYR7Y/action/storage_attestation","attest_author":"https://pith.science/pith/E6S44KXPKU2POIFRZIK76TYR7Y/action/author_attestation","sign_citation":"https://pith.science/pith/E6S44KXPKU2POIFRZIK76TYR7Y/action/citation_signature","submit_replication":"https://pith.science/pith/E6S44KXPKU2POIFRZIK76TYR7Y/action/replication_record"}},"created_at":"2026-07-05T06:20:44.748482+00:00","updated_at":"2026-07-05T06:20:44.748482+00:00"}