{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:L6PL36PLD4OGIT6FJ4Q3QBSMG2","short_pith_number":"pith:L6PL36PL","schema_version":"1.0","canonical_sha256":"5f9ebdf9eb1f1c644fc54f21b8064c36a38bb499596c0cbedb89bb262d74c1ee","source":{"kind":"arxiv","id":"2301.12867","version":4},"attestation_state":"computed","paper":{"title":"Red teaming ChatGPT via Jailbreaking: Bias, Robustness, Reliability and Toxicity","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SE"],"primary_cat":"cs.CL","authors_text":"Chunyang Chen, Terry Yue Zhuo, Yujin Huang, Zhenchang Xing","submitted_at":"2023-01-30T13:20:48Z","abstract_excerpt":"Recent breakthroughs in natural language processing (NLP) have permitted the synthesis and comprehension of coherent text in an open-ended way, therefore translating the theoretical algorithms into practical applications. The large language models (LLMs) have significantly impacted businesses such as report summarization software and copywriters. Observations indicate, however, that LLMs may exhibit social prejudice and toxicity, posing ethical and societal dangers of consequences resulting from irresponsibility. Large-scale benchmarks for accountable LLMs should consequently be developed. Alt"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2301.12867","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-01-30T13:20:48Z","cross_cats_sorted":["cs.SE"],"title_canon_sha256":"be3a4f55e9295c907c91fd36da6837e5a7fb51f7aa9ff26fe811c37695f14cd0","abstract_canon_sha256":"3e5a2cdf832c47bf1eca227eccd61530957d5c05b8967282ea1a711090a99141"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:14:40.092806Z","signature_b64":"sRuMH9TLuYBX0OWRDb6bGqFOIyJnJ9h3HiMXyYq6G8kcpUYEWNym6qXvVVkzQKy94G9dhKncMxBrFl8fatCfDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5f9ebdf9eb1f1c644fc54f21b8064c36a38bb499596c0cbedb89bb262d74c1ee","last_reissued_at":"2026-07-05T06:14:40.092392Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:14:40.092392Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Red teaming ChatGPT via Jailbreaking: Bias, Robustness, Reliability and Toxicity","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SE"],"primary_cat":"cs.CL","authors_text":"Chunyang Chen, Terry Yue Zhuo, Yujin Huang, Zhenchang Xing","submitted_at":"2023-01-30T13:20:48Z","abstract_excerpt":"Recent breakthroughs in natural language processing (NLP) have permitted the synthesis and comprehension of coherent text in an open-ended way, therefore translating the theoretical algorithms into practical applications. The large language models (LLMs) have significantly impacted businesses such as report summarization software and copywriters. Observations indicate, however, that LLMs may exhibit social prejudice and toxicity, posing ethical and societal dangers of consequences resulting from irresponsibility. Large-scale benchmarks for accountable LLMs should consequently be developed. Alt"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2301.12867","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2301.12867/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2301.12867","created_at":"2026-07-05T06:14:40.092452+00:00"},{"alias_kind":"arxiv_version","alias_value":"2301.12867v4","created_at":"2026-07-05T06:14:40.092452+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2301.12867","created_at":"2026-07-05T06:14:40.092452+00:00"},{"alias_kind":"pith_short_12","alias_value":"L6PL36PLD4OG","created_at":"2026-07-05T06:14:40.092452+00:00"},{"alias_kind":"pith_short_16","alias_value":"L6PL36PLD4OGIT6F","created_at":"2026-07-05T06:14:40.092452+00:00"},{"alias_kind":"pith_short_8","alias_value":"L6PL36PL","created_at":"2026-07-05T06:14:40.092452+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07480","citing_title":"Biased or Personalized? The Impact of Personal Information on AI-driven Development","ref_index":81,"is_internal_anchor":true},{"citing_arxiv_id":"2606.25380","citing_title":"A Survey of Toxicity Detection and Mitigation Strategies for Multilingual Language Models","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19887","citing_title":"FinRED: An Expert-Guided Benchmark Generation and Evaluation Framework for Financial LLM Red-Teaming","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12422","citing_title":"Creating and Evaluating K-12 GenAI Assessment Graders Through Context Engineering","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30783","citing_title":"Security--Fidelity Tradeoffs: The Hidden Cost of Prompt Injection Defense","ref_index":110,"is_internal_anchor":false},{"citing_arxiv_id":"2403.03920","citing_title":"Enhancing Instructional Quality: Leveraging Computer-Assisted Textual Analysis to Generate In-Depth Insights from Educational Artifacts","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2410.15362","citing_title":"Faster-GCG: Efficient Discrete Optimization Jailbreak Attacks against Aligned Large Language Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2503.19444","citing_title":"AI Failures in the Eyes of the Downstream Developer: A First Look at Concerns, Practices, and Challenges","ref_index":124,"is_internal_anchor":false},{"citing_arxiv_id":"2308.05374","citing_title":"Trustworthy LLMs: a Survey and Guideline for Evaluating Large Language Models' Alignment","ref_index":219,"is_internal_anchor":false},{"citing_arxiv_id":"2310.02446","citing_title":"Low-Resource Languages Jailbreak GPT-4","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2402.19173","citing_title":"StarCoder 2 and The Stack v2: The Next Generation","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2310.04451","citing_title":"AutoDAN: Generating Stealthy Jailbreak Prompts on Aligned Large Language Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22089","citing_title":"Ethics Testing: Proactive Identification of Generative AI System Harms","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20677","citing_title":"Intersectional Fairness in Large Language Models","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18946","citing_title":"Reasoning Structure Matters for Safety Alignment of Reasoning Models","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/L6PL36PLD4OGIT6FJ4Q3QBSMG2","json":"https://pith.science/pith/L6PL36PLD4OGIT6FJ4Q3QBSMG2.json","graph_json":"https://pith.science/api/pith-number/L6PL36PLD4OGIT6FJ4Q3QBSMG2/graph.json","events_json":"https://pith.science/api/pith-number/L6PL36PLD4OGIT6FJ4Q3QBSMG2/events.json","paper":"https://pith.science/paper/L6PL36PL"},"agent_actions":{"view_html":"https://pith.science/pith/L6PL36PLD4OGIT6FJ4Q3QBSMG2","download_json":"https://pith.science/pith/L6PL36PLD4OGIT6FJ4Q3QBSMG2.json","view_paper":"https://pith.science/paper/L6PL36PL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2301.12867&json=true","fetch_graph":"https://pith.science/api/pith-number/L6PL36PLD4OGIT6FJ4Q3QBSMG2/graph.json","fetch_events":"https://pith.science/api/pith-number/L6PL36PLD4OGIT6FJ4Q3QBSMG2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/L6PL36PLD4OGIT6FJ4Q3QBSMG2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/L6PL36PLD4OGIT6FJ4Q3QBSMG2/action/storage_attestation","attest_author":"https://pith.science/pith/L6PL36PLD4OGIT6FJ4Q3QBSMG2/action/author_attestation","sign_citation":"https://pith.science/pith/L6PL36PLD4OGIT6FJ4Q3QBSMG2/action/citation_signature","submit_replication":"https://pith.science/pith/L6PL36PLD4OGIT6FJ4Q3QBSMG2/action/replication_record"}},"created_at":"2026-07-05T06:14:40.092452+00:00","updated_at":"2026-07-05T06:14:40.092452+00:00"}