{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:PNLVJYO36Z4J42A4M3FVFWRQEV","short_pith_number":"pith:PNLVJYO3","schema_version":"1.0","canonical_sha256":"7b5754e1dbf6789e681c66cb52da302552a21a0455fccf7a3f696382d5bd3787","source":{"kind":"arxiv","id":"2506.15253","version":1},"attestation_state":"computed","paper":{"title":"RAS-Eval: A Comprehensive Benchmark for Security Evaluation of LLM Agents in Real-World Environments","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Dongxia Wang, Xiaohan Yuan, Yuchuan Fu","submitted_at":"2025-06-18T08:30:36Z","abstract_excerpt":"The rapid deployment of Large language model (LLM) agents in critical domains like healthcare and finance necessitates robust security frameworks. To address the absence of standardized evaluation benchmarks for these agents in dynamic environments, we introduce RAS-Eval, a comprehensive security benchmark supporting both simulated and real-world tool execution. RAS-Eval comprises 80 test cases and 3,802 attack tasks mapped to 11 Common Weakness Enumeration (CWE) categories, with tools implemented in JSON, LangGraph, and Model Context Protocol (MCP) formats. We evaluate 6 state-of-the-art LLMs"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.15253","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2025-06-18T08:30:36Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"08a3e17e8222728b9c725635bb5df9d9a48a01c4a8fa410109c50b9b83615e6a","abstract_canon_sha256":"a4a41e480c3b0848fe4b23168fe1bee6e6547e1ce93787f29cf735b657cfd6b8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:23:40.213236Z","signature_b64":"PbeLqM7qvIjdtRBUVN1fY0jDYhRmwb6bdhL1CSXYBEYNG/BnUwtcY7gsqu8M/ptURB0KPihA2JqvryQGZFY6Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7b5754e1dbf6789e681c66cb52da302552a21a0455fccf7a3f696382d5bd3787","last_reissued_at":"2026-07-05T11:23:40.212820Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:23:40.212820Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RAS-Eval: A Comprehensive Benchmark for Security Evaluation of LLM Agents in Real-World Environments","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Dongxia Wang, Xiaohan Yuan, Yuchuan Fu","submitted_at":"2025-06-18T08:30:36Z","abstract_excerpt":"The rapid deployment of Large language model (LLM) agents in critical domains like healthcare and finance necessitates robust security frameworks. To address the absence of standardized evaluation benchmarks for these agents in dynamic environments, we introduce RAS-Eval, a comprehensive security benchmark supporting both simulated and real-world tool execution. RAS-Eval comprises 80 test cases and 3,802 attack tasks mapped to 11 Common Weakness Enumeration (CWE) categories, with tools implemented in JSON, LangGraph, and Model Context Protocol (MCP) formats. We evaluate 6 state-of-the-art LLMs"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.15253","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.15253/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.15253","created_at":"2026-07-05T11:23:40.212880+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.15253v1","created_at":"2026-07-05T11:23:40.212880+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.15253","created_at":"2026-07-05T11:23:40.212880+00:00"},{"alias_kind":"pith_short_12","alias_value":"PNLVJYO36Z4J","created_at":"2026-07-05T11:23:40.212880+00:00"},{"alias_kind":"pith_short_16","alias_value":"PNLVJYO36Z4J42A4","created_at":"2026-07-05T11:23:40.212880+00:00"},{"alias_kind":"pith_short_8","alias_value":"PNLVJYO3","created_at":"2026-07-05T11:23:40.212880+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.24069","citing_title":"When the Manual Lies: A Realistic Benchmark to Evaluate MCP Poisoning Attacks for LLM Agents","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10749","citing_title":"Toward Secure LLM Agents: Threat Surfaces, Attacks, Defenses, and Evaluation","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11053","citing_title":"Content-Aware Attack Detection in LLM Agent Tool-Call Traffic: An Empirical Study of Features, Architectures, and Evaluation Protocols","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16282","citing_title":"Taxonomy and Consistency Analysis of Safety Benchmarks for AI Agents","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17380","citing_title":"ADR: An Agentic Detection System for Enterprise Agentic AI Security","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14859","citing_title":"Do Coding Agents Understand Least-Privilege Authorization?","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2603.28166","citing_title":"Evaluating Privilege Usage of Agents with Real-World Tools","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11053","citing_title":"Content-Aware Attack Detection in LLM Agent Tool-Call Traffic: An Empirical Study of Features, Architectures, and Evaluation Protocols","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11053","citing_title":"Content-Aware Attack Detection in LLM Agent Tool-Call Traffic: An Empirical Study of Features, Architectures, and Evaluation Protocols","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PNLVJYO36Z4J42A4M3FVFWRQEV","json":"https://pith.science/pith/PNLVJYO36Z4J42A4M3FVFWRQEV.json","graph_json":"https://pith.science/api/pith-number/PNLVJYO36Z4J42A4M3FVFWRQEV/graph.json","events_json":"https://pith.science/api/pith-number/PNLVJYO36Z4J42A4M3FVFWRQEV/events.json","paper":"https://pith.science/paper/PNLVJYO3"},"agent_actions":{"view_html":"https://pith.science/pith/PNLVJYO36Z4J42A4M3FVFWRQEV","download_json":"https://pith.science/pith/PNLVJYO36Z4J42A4M3FVFWRQEV.json","view_paper":"https://pith.science/paper/PNLVJYO3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.15253&json=true","fetch_graph":"https://pith.science/api/pith-number/PNLVJYO36Z4J42A4M3FVFWRQEV/graph.json","fetch_events":"https://pith.science/api/pith-number/PNLVJYO36Z4J42A4M3FVFWRQEV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PNLVJYO36Z4J42A4M3FVFWRQEV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PNLVJYO36Z4J42A4M3FVFWRQEV/action/storage_attestation","attest_author":"https://pith.science/pith/PNLVJYO36Z4J42A4M3FVFWRQEV/action/author_attestation","sign_citation":"https://pith.science/pith/PNLVJYO36Z4J42A4M3FVFWRQEV/action/citation_signature","submit_replication":"https://pith.science/pith/PNLVJYO36Z4J42A4M3FVFWRQEV/action/replication_record"}},"created_at":"2026-07-05T11:23:40.212880+00:00","updated_at":"2026-07-05T11:23:40.212880+00:00"}