{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:5DMEZKUIDNUJG3HGOCDHQNMBIU","short_pith_number":"pith:5DMEZKUI","schema_version":"1.0","canonical_sha256":"e8d84caa881b68936ce67086783581452bef4d19d24733dd1226a70c825ac783","source":{"kind":"arxiv","id":"2505.13862","version":3},"attestation_state":"computed","paper":{"title":"PandaGuard: Systematic Evaluation of LLM Safety against Jailbreaking Attacks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CR","authors_text":"Dongcheng Zhao, Guobin Shen, Haibo Tong, Jihang Wang, Jindong Li, Linghao Feng, Sicheng Shen, Xiang He, Xiang Zheng, Yiting Dong, Yi Zeng","submitted_at":"2025-05-20T03:14:57Z","abstract_excerpt":"Large language models (LLMs) have achieved remarkable capabilities but remain vulnerable to adversarial prompts known as jailbreaks, which can bypass safety alignment and elicit harmful outputs. Despite growing efforts in LLM safety research, existing evaluations are often fragmented, focused on isolated attack or defense techniques, and lack systematic, reproducible analysis. In this work, we introduce PandaGuard, a unified and modular framework that models LLM jailbreak safety as a multi-agent system comprising attackers, defenders, and judges. Our framework implements 19 attack methods and "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.13862","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2025-05-20T03:14:57Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"a8453c9ace476fe53e5097dae99d63f6bc6d56307efca04395d6c29d4af66d9e","abstract_canon_sha256":"1fb6db44820cce2e51e55ccbcfc3ff663df6b3201266c8a26ac62e6622bcd5ce"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:09:47.422491Z","signature_b64":"tjwZ0ITEbzLSYzpnyCUADPEpdabAvUg6nALGR+gUllRnd53o4N9lqjVH4YQPhQJ9y/doqewOmW4c4yObwBPuBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e8d84caa881b68936ce67086783581452bef4d19d24733dd1226a70c825ac783","last_reissued_at":"2026-07-05T11:09:47.421874Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:09:47.421874Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PandaGuard: Systematic Evaluation of LLM Safety against Jailbreaking Attacks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CR","authors_text":"Dongcheng Zhao, Guobin Shen, Haibo Tong, Jihang Wang, Jindong Li, Linghao Feng, Sicheng Shen, Xiang He, Xiang Zheng, Yiting Dong, Yi Zeng","submitted_at":"2025-05-20T03:14:57Z","abstract_excerpt":"Large language models (LLMs) have achieved remarkable capabilities but remain vulnerable to adversarial prompts known as jailbreaks, which can bypass safety alignment and elicit harmful outputs. Despite growing efforts in LLM safety research, existing evaluations are often fragmented, focused on isolated attack or defense techniques, and lack systematic, reproducible analysis. In this work, we introduce PandaGuard, a unified and modular framework that models LLM jailbreak safety as a multi-agent system comprising attackers, defenders, and judges. Our framework implements 19 attack methods and "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.13862","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.13862/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.13862","created_at":"2026-07-05T11:09:47.421961+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.13862v3","created_at":"2026-07-05T11:09:47.421961+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.13862","created_at":"2026-07-05T11:09:47.421961+00:00"},{"alias_kind":"pith_short_12","alias_value":"5DMEZKUIDNUJ","created_at":"2026-07-05T11:09:47.421961+00:00"},{"alias_kind":"pith_short_16","alias_value":"5DMEZKUIDNUJG3HG","created_at":"2026-07-05T11:09:47.421961+00:00"},{"alias_kind":"pith_short_8","alias_value":"5DMEZKUI","created_at":"2026-07-05T11:09:47.421961+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.09708","citing_title":"Beyond I'm Sorry, I Can't: Dissecting Large Language Model Refusal","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2512.20677","citing_title":"Learning-Based Automated Adversarial Red-Teaming for Robustness Evaluation of Large Language Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05058","citing_title":"SoK: Robustness in Large Language Models against Jailbreak Attacks","ref_index":71,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5DMEZKUIDNUJG3HGOCDHQNMBIU","json":"https://pith.science/pith/5DMEZKUIDNUJG3HGOCDHQNMBIU.json","graph_json":"https://pith.science/api/pith-number/5DMEZKUIDNUJG3HGOCDHQNMBIU/graph.json","events_json":"https://pith.science/api/pith-number/5DMEZKUIDNUJG3HGOCDHQNMBIU/events.json","paper":"https://pith.science/paper/5DMEZKUI"},"agent_actions":{"view_html":"https://pith.science/pith/5DMEZKUIDNUJG3HGOCDHQNMBIU","download_json":"https://pith.science/pith/5DMEZKUIDNUJG3HGOCDHQNMBIU.json","view_paper":"https://pith.science/paper/5DMEZKUI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.13862&json=true","fetch_graph":"https://pith.science/api/pith-number/5DMEZKUIDNUJG3HGOCDHQNMBIU/graph.json","fetch_events":"https://pith.science/api/pith-number/5DMEZKUIDNUJG3HGOCDHQNMBIU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5DMEZKUIDNUJG3HGOCDHQNMBIU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5DMEZKUIDNUJG3HGOCDHQNMBIU/action/storage_attestation","attest_author":"https://pith.science/pith/5DMEZKUIDNUJG3HGOCDHQNMBIU/action/author_attestation","sign_citation":"https://pith.science/pith/5DMEZKUIDNUJG3HGOCDHQNMBIU/action/citation_signature","submit_replication":"https://pith.science/pith/5DMEZKUIDNUJG3HGOCDHQNMBIU/action/replication_record"}},"created_at":"2026-07-05T11:09:47.421961+00:00","updated_at":"2026-07-05T11:09:47.421961+00:00"}