{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:J5VOWRAVPSNFNVLKMWUKY2HAJO","short_pith_number":"pith:J5VOWRAV","schema_version":"1.0","canonical_sha256":"4f6aeb44157c9a56d56a65a8ac68e04b9e55d7c8837a5e2ed66b6588e02dc371","source":{"kind":"arxiv","id":"2501.14654","version":2},"attestation_state":"computed","paper":{"title":"MedAgentBench: A Realistic Virtual EHR Environment to Benchmark Medical LLM Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.MA"],"primary_cat":"cs.LG","authors_text":"Andrew Y. Ng, Danny Park, Gloria Geng, James Zou, Jonathan H. Chen, Kameron C. Black, Yixing Jiang","submitted_at":"2025-01-24T17:21:01Z","abstract_excerpt":"Recent large language models (LLMs) have demonstrated significant advancements, particularly in their ability to serve as agents thereby surpassing their traditional role as chatbots. These agents can leverage their planning and tool utilization capabilities to address tasks specified at a high level. However, a standardized dataset to benchmark the agent capabilities of LLMs in medical applications is currently lacking, making the evaluation of LLMs on complex tasks in interactive healthcare environments challenging. To address this gap, we introduce MedAgentBench, a broad evaluation suite de"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.14654","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-24T17:21:01Z","cross_cats_sorted":["cs.AI","cs.MA"],"title_canon_sha256":"20c83a8076cea3be3353f63d965a903bc684e93acb3f991118abe78547f70add","abstract_canon_sha256":"3b481554c0a8b91ed4c7253546152180ba5e5ba5250c43b9b98395a10cc81f44"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:13:07.595024Z","signature_b64":"rfyEDVteqicSobXS/jjo7naDNTQU2qkYkIKDZA2a5dobE2ss8FCW2l0BmLjdaehRXjJINDPRoYx3Hh9tYUD3AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4f6aeb44157c9a56d56a65a8ac68e04b9e55d7c8837a5e2ed66b6588e02dc371","last_reissued_at":"2026-07-05T10:13:07.594636Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:13:07.594636Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MedAgentBench: A Realistic Virtual EHR Environment to Benchmark Medical LLM Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.MA"],"primary_cat":"cs.LG","authors_text":"Andrew Y. Ng, Danny Park, Gloria Geng, James Zou, Jonathan H. Chen, Kameron C. Black, Yixing Jiang","submitted_at":"2025-01-24T17:21:01Z","abstract_excerpt":"Recent large language models (LLMs) have demonstrated significant advancements, particularly in their ability to serve as agents thereby surpassing their traditional role as chatbots. These agents can leverage their planning and tool utilization capabilities to address tasks specified at a high level. However, a standardized dataset to benchmark the agent capabilities of LLMs in medical applications is currently lacking, making the evaluation of LLMs on complex tasks in interactive healthcare environments challenging. To address this gap, we introduce MedAgentBench, a broad evaluation suite de"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.14654","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.14654/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.14654","created_at":"2026-07-05T10:13:07.594682+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.14654v2","created_at":"2026-07-05T10:13:07.594682+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.14654","created_at":"2026-07-05T10:13:07.594682+00:00"},{"alias_kind":"pith_short_12","alias_value":"J5VOWRAVPSNF","created_at":"2026-07-05T10:13:07.594682+00:00"},{"alias_kind":"pith_short_16","alias_value":"J5VOWRAVPSNFNVLK","created_at":"2026-07-05T10:13:07.594682+00:00"},{"alias_kind":"pith_short_8","alias_value":"J5VOWRAV","created_at":"2026-07-05T10:13:07.594682+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26346","citing_title":"How Do Tool-Augmented LLM Agents Perform on Real-World Energy Analytics Tasks?","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.16723","citing_title":"AgentFairBench: Do LLM Agents Discriminate When They Act?","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13608","citing_title":"AgentBeats: Agentifying Agent Assessment for Openness, Standardization, and Reproducibility","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12736","citing_title":"Benchmarking AI Agents for Addressing Scientific Challenges Across Scales","ref_index":145,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11740","citing_title":"UniReason-Med: A Shared Grounded Reasoning Interface for 2D-to-3D Transfer in Medical VQA","ref_index":157,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01961","citing_title":"AutoMedBench: Towards Medical AutoResearch with Agentic AI Models","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06177","citing_title":"BioMedArena: An Open-source Toolkit for Building and Evaluating Biomedical Deep Research Agents","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28900","citing_title":"MedEvoEval: Evaluating Continual Evolution of Doctor Agents through Simulated Clinical Episodes","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29602","citing_title":"An Empirical Evaluation of Prompt Injection Vulnerabilities in Large Language Models Across Multilingual and Obfuscated Attack Scenarios","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12191","citing_title":"Agentic Environment Engineering for Large Language Models: A Survey of Environment Modeling, Synthesis, Evaluation, and Application","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12291","citing_title":"Measuring Epistemic Resilience of LLMs Under Misleading Medical Context","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23262","citing_title":"Design and Report Benchmarks for Knowledge Work","ref_index":127,"is_internal_anchor":false},{"citing_arxiv_id":"2503.21460","citing_title":"Large Language Model Agent: A Survey on Methodology, Applications and Challenges","ref_index":136,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18768","citing_title":"ClinQueryAgent: A Conversational Agent for Population Health Management","ref_index":148,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06177","citing_title":"BioMedArena: An Open-source Toolkit for Building and Evaluating Biomedical Deep Research Agents","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21255","citing_title":"When Agents Look the Same: Quantifying Distillation-Induced Similarity in Tool-Use Behaviors","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/J5VOWRAVPSNFNVLKMWUKY2HAJO","json":"https://pith.science/pith/J5VOWRAVPSNFNVLKMWUKY2HAJO.json","graph_json":"https://pith.science/api/pith-number/J5VOWRAVPSNFNVLKMWUKY2HAJO/graph.json","events_json":"https://pith.science/api/pith-number/J5VOWRAVPSNFNVLKMWUKY2HAJO/events.json","paper":"https://pith.science/paper/J5VOWRAV"},"agent_actions":{"view_html":"https://pith.science/pith/J5VOWRAVPSNFNVLKMWUKY2HAJO","download_json":"https://pith.science/pith/J5VOWRAVPSNFNVLKMWUKY2HAJO.json","view_paper":"https://pith.science/paper/J5VOWRAV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.14654&json=true","fetch_graph":"https://pith.science/api/pith-number/J5VOWRAVPSNFNVLKMWUKY2HAJO/graph.json","fetch_events":"https://pith.science/api/pith-number/J5VOWRAVPSNFNVLKMWUKY2HAJO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/J5VOWRAVPSNFNVLKMWUKY2HAJO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/J5VOWRAVPSNFNVLKMWUKY2HAJO/action/storage_attestation","attest_author":"https://pith.science/pith/J5VOWRAVPSNFNVLKMWUKY2HAJO/action/author_attestation","sign_citation":"https://pith.science/pith/J5VOWRAVPSNFNVLKMWUKY2HAJO/action/citation_signature","submit_replication":"https://pith.science/pith/J5VOWRAVPSNFNVLKMWUKY2HAJO/action/replication_record"}},"created_at":"2026-07-05T10:13:07.594682+00:00","updated_at":"2026-07-05T10:13:07.594682+00:00"}