{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:QP3GIW637IHRRNZASM6IWC3SS4","short_pith_number":"pith:QP3GIW63","schema_version":"1.0","canonical_sha256":"83f6645bdbfa0f18b720933c8b0b7297144950234ff9932afdced0d8b8f75347","source":{"kind":"arxiv","id":"2312.04127","version":2},"attestation_state":"computed","paper":{"title":"Analyzing the Inherent Response Tendency of LLMs: Real-World Instructions-Driven Jailbreak","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bing Qin, Ming Ma, Sendong Zhao, Yanrui Du, Yuhan Chen","submitted_at":"2023-12-07T08:29:58Z","abstract_excerpt":"Extensive work has been devoted to improving the safety mechanism of Large Language Models (LLMs). However, LLMs still tend to generate harmful responses when faced with malicious instructions, a phenomenon referred to as \"Jailbreak Attack\". In our research, we introduce a novel automatic jailbreak method RADIAL, which bypasses the security mechanism by amplifying the potential of LLMs to generate affirmation responses. The jailbreak idea of our method is \"Inherent Response Tendency Analysis\" which identifies real-world instructions that can inherently induce LLMs to generate affirmation respo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.04127","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-12-07T08:29:58Z","cross_cats_sorted":[],"title_canon_sha256":"0f4ff6a9e0357f3fedd1b8048454a73a39741ecd213f64f1373c3f159bf272c0","abstract_canon_sha256":"fd25da4677beb71ff50ea43b643afa3faf881d8d076eb98f05dfe1fb09124b89"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:48:33.116369Z","signature_b64":"rG2iAWEQl8eq5APt/lhSjbe1ezOuFoTgpO6gwpDEw4MuOl3DZ3jRgeAUUoXJ6LXJ5WSjl+TNUk9dWaFO1w/0AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"83f6645bdbfa0f18b720933c8b0b7297144950234ff9932afdced0d8b8f75347","last_reissued_at":"2026-07-05T07:48:33.115897Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:48:33.115897Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Analyzing the Inherent Response Tendency of LLMs: Real-World Instructions-Driven Jailbreak","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bing Qin, Ming Ma, Sendong Zhao, Yanrui Du, Yuhan Chen","submitted_at":"2023-12-07T08:29:58Z","abstract_excerpt":"Extensive work has been devoted to improving the safety mechanism of Large Language Models (LLMs). However, LLMs still tend to generate harmful responses when faced with malicious instructions, a phenomenon referred to as \"Jailbreak Attack\". In our research, we introduce a novel automatic jailbreak method RADIAL, which bypasses the security mechanism by amplifying the potential of LLMs to generate affirmation responses. The jailbreak idea of our method is \"Inherent Response Tendency Analysis\" which identifies real-world instructions that can inherently induce LLMs to generate affirmation respo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.04127","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.04127/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.04127","created_at":"2026-07-05T07:48:33.115958+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.04127v2","created_at":"2026-07-05T07:48:33.115958+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.04127","created_at":"2026-07-05T07:48:33.115958+00:00"},{"alias_kind":"pith_short_12","alias_value":"QP3GIW637IHR","created_at":"2026-07-05T07:48:33.115958+00:00"},{"alias_kind":"pith_short_16","alias_value":"QP3GIW637IHRRNZA","created_at":"2026-07-05T07:48:33.115958+00:00"},{"alias_kind":"pith_short_8","alias_value":"QP3GIW63","created_at":"2026-07-05T07:48:33.115958+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2407.04295","citing_title":"Jailbreak Attacks and Defenses Against Large Language Models: A Survey","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QP3GIW637IHRRNZASM6IWC3SS4","json":"https://pith.science/pith/QP3GIW637IHRRNZASM6IWC3SS4.json","graph_json":"https://pith.science/api/pith-number/QP3GIW637IHRRNZASM6IWC3SS4/graph.json","events_json":"https://pith.science/api/pith-number/QP3GIW637IHRRNZASM6IWC3SS4/events.json","paper":"https://pith.science/paper/QP3GIW63"},"agent_actions":{"view_html":"https://pith.science/pith/QP3GIW637IHRRNZASM6IWC3SS4","download_json":"https://pith.science/pith/QP3GIW637IHRRNZASM6IWC3SS4.json","view_paper":"https://pith.science/paper/QP3GIW63","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.04127&json=true","fetch_graph":"https://pith.science/api/pith-number/QP3GIW637IHRRNZASM6IWC3SS4/graph.json","fetch_events":"https://pith.science/api/pith-number/QP3GIW637IHRRNZASM6IWC3SS4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QP3GIW637IHRRNZASM6IWC3SS4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QP3GIW637IHRRNZASM6IWC3SS4/action/storage_attestation","attest_author":"https://pith.science/pith/QP3GIW637IHRRNZASM6IWC3SS4/action/author_attestation","sign_citation":"https://pith.science/pith/QP3GIW637IHRRNZASM6IWC3SS4/action/citation_signature","submit_replication":"https://pith.science/pith/QP3GIW637IHRRNZASM6IWC3SS4/action/replication_record"}},"created_at":"2026-07-05T07:48:33.115958+00:00","updated_at":"2026-07-05T07:48:33.115958+00:00"}