{"as_of":"2026-08-07T11:37:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:8d80e3fe778e800c8ebb3650fd178a7e0b2f58d21b52552dfbae4d8fe5e789ce","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":20,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":20,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":20,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":20,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T16:42:29.834880Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":4,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.14654","last_updated":"2025-02-12T05:32:07Z","snapshot_observed_at":"2026-07-06T20:25:40.715105Z","submitted_at":"2025-01-24T17:21:01Z","title":"MedAgentBench: A Realistic Virtual EHR Environment to Benchmark Medical LLM Agents","version":2},"cited_work":{"arxiv_id":"2501.14654","doi":"10.48550/arxiv.2501.14654","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.14654","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Geng, Gloria and Park, Danny and Zou, James and Ng, Andrew Y","venue":"ArXiv.org","work_id":"52ee24a4-e815-4788-8c57-f947197cc09e","year":2025},"citing_paper":{"arxiv_id":"2503.21460","last_updated":"2025-03-27T12:50:17Z","snapshot_observed_at":"2026-07-06T20:59:35.694800Z","submitted_at":"2025-03-27T12:50:17Z","title":"Large Language Model Agent: A Survey on Methodology, Applications and Challenges","version":1},"reference_index":136,"source":"pdf_text","source_observed_at":"2026-05-22T21:51:34.309870Z"},"links":{"cited_paper":"/paper/2501.14654","citing_paper":"/paper/2503.21460"},"observation_digest":"sha256:1a6787462a13cd21ca6c5c8656b7f827ddb35006962ad68bf57b59d7e920b0fd","observation_id":"0b8e6533-2940-4938-bc2b-30cd95ddd63f","resolution":{"observed_at":"2026-05-22T21:52:10.740837Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14654","last_updated":"2025-02-12T05:32:07Z","snapshot_observed_at":"2026-07-06T20:25:40.715105Z","submitted_at":"2025-01-24T17:21:01Z","title":"MedAgentBench: A Realistic Virtual EHR Environment to Benchmark Medical LLM Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.14654","snapshot_observed_at":"2026-08-06T16:42:29.834880Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.12774","last_updated":"2025-07-17T04:31:55Z","snapshot_observed_at":"2026-08-06T16:36:30.473430Z","submitted_at":"2025-07-17T04:31:55Z","title":"A Comprehensive Survey of Electronic Health Record Modeling: From Deep Learning Approaches to Large Language Models","version":1},"reference_index":118,"source":"pdf_text","source_observed_at":"2026-08-06T16:42:29.834880Z"},"links":{"cited_paper":"/paper/2501.14654","citing_paper":"/paper/2507.12774"},"observation_digest":"sha256:3afa59fb62f6765b973cd7d5f6828b1a1d7cedea4482fb4573c9ff7802959777","observation_id":"766ac164-8ae7-487c-9284-875414760586","resolution":{"observed_at":"2026-08-06T16:42:29.834880Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14654","last_updated":"2025-02-12T05:32:07Z","snapshot_observed_at":"2026-07-06T20:25:40.715105Z","submitted_at":"2025-01-24T17:21:01Z","title":"MedAgentBench: A Realistic Virtual EHR Environment to Benchmark Medical LLM Agents","version":2},"cited_work":{"arxiv_id":"2501.14654","doi":"10.48550/arxiv.2501.14654","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.14654","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Geng, Gloria and Park, Danny and Zou, James and Ng, Andrew Y","venue":"ArXiv.org","work_id":"52ee24a4-e815-4788-8c57-f947197cc09e","year":2025},"citing_paper":{"arxiv_id":"2604.21255","last_updated":"2026-04-23T03:48:56Z","snapshot_observed_at":"2026-07-06T23:07:51.979267Z","submitted_at":"2026-04-23T03:48:56Z","title":"When Agents Look the Same: Quantifying Distillation-Induced Similarity in Tool-Use Behaviors","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-09T22:07:58.614654Z"},"links":{"cited_paper":"/paper/2501.14654","citing_paper":"/paper/2604.21255"},"observation_digest":"sha256:ce4ff0a3eccb4fb3d5d1522a56aaa769cf6bc7497c0899527a708e522bd22b86","observation_id":"e9af4b11-6ba5-424b-98b2-7eec7f0c8735","resolution":{"observed_at":"2026-05-11T14:16:18.019393Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14654","last_updated":"2025-02-12T05:32:07Z","snapshot_observed_at":"2026-07-06T20:25:40.715105Z","submitted_at":"2025-01-24T17:21:01Z","title":"MedAgentBench: A Realistic Virtual EHR Environment to Benchmark Medical LLM Agents","version":2},"cited_work":{"arxiv_id":"2501.14654","doi":"10.48550/arxiv.2501.14654","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.14654","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Geng, Gloria and Park, Danny and Zou, James and Ng, Andrew Y","venue":"ArXiv.org","work_id":"52ee24a4-e815-4788-8c57-f947197cc09e","year":2025},"citing_paper":{"arxiv_id":"2605.06177","last_updated":"2026-06-23T17:07:51Z","snapshot_observed_at":"2026-07-06T23:18:41.400741Z","submitted_at":"2026-05-07T12:57:18Z","title":"BioMedArena: An Open-source Toolkit for Building and Evaluating Biomedical Deep Research Agents","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-08T10:22:34.260242Z"},"links":{"cited_paper":"/paper/2501.14654","citing_paper":"/paper/2605.06177"},"observation_digest":"sha256:7ab358e7524d6974139a84f9a16147538bf2019051f99f0924a24fcc387407c4","observation_id":"5c8bd682-a9b2-4fd6-84c2-078c266d92fd","resolution":{"observed_at":"2026-05-11T20:06:08.854680Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14654","last_updated":"2025-02-12T05:32:07Z","snapshot_observed_at":"2026-07-06T20:25:40.715105Z","submitted_at":"2025-01-24T17:21:01Z","title":"MedAgentBench: A Realistic Virtual EHR Environment to Benchmark Medical LLM Agents","version":2},"cited_work":{"arxiv_id":"2501.14654","doi":"10.48550/arxiv.2501.14654","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.14654","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Geng, Gloria and Park, Danny and Zou, James and Ng, Andrew Y","venue":"ArXiv.org","work_id":"52ee24a4-e815-4788-8c57-f947197cc09e","year":2025},"citing_paper":{"arxiv_id":"2605.06177","last_updated":"2026-06-23T17:07:51Z","snapshot_observed_at":"2026-07-06T23:18:41.400741Z","submitted_at":"2026-05-07T12:57:18Z","title":"BioMedArena: An Open-source Toolkit for Building and Evaluating Biomedical Deep Research Agents","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-30T23:33:52.182432Z"},"links":{"cited_paper":"/paper/2501.14654","citing_paper":"/paper/2605.06177"},"observation_digest":"sha256:1107547f8490ea74f16165ab148b64b36ad58127576e12460bea072feca94673","observation_id":"f9ee0f25-79ba-4d32-86d2-bf01797963ec","resolution":{"observed_at":"2026-06-30T23:35:07.124201Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14654","last_updated":"2025-02-12T05:32:07Z","snapshot_observed_at":"2026-07-06T20:25:40.715105Z","submitted_at":"2025-01-24T17:21:01Z","title":"MedAgentBench: A Realistic Virtual EHR Environment to Benchmark Medical LLM Agents","version":2},"cited_work":{"arxiv_id":"2501.14654","doi":"10.48550/arxiv.2501.14654","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.14654","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Geng, Gloria and Park, Danny and Zou, James and Ng, Andrew Y","venue":"ArXiv.org","work_id":"52ee24a4-e815-4788-8c57-f947197cc09e","year":2025},"citing_paper":{"arxiv_id":"2605.18768","last_updated":"2026-04-13T10:07:03Z","snapshot_observed_at":"2026-08-01T02:49:14.468592Z","submitted_at":"2026-04-13T10:07:03Z","title":"ClinQueryAgent: A Conversational Agent for Population Health Management","version":1},"reference_index":148,"source":"arxiv_source","source_observed_at":"2026-05-21T01:31:07.031424Z"},"links":{"cited_paper":"/paper/2501.14654","citing_paper":"/paper/2605.18768"},"observation_digest":"sha256:8497f3447a5467dba27ffb9e80ea4d1049c89ab2604aadfa22b5dce7ad996c3e","observation_id":"a4f7beb2-ec53-4c51-983b-a2703a7ce921","resolution":{"observed_at":"2026-05-21T01:33:55.639019Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14654","last_updated":"2025-02-12T05:32:07Z","snapshot_observed_at":"2026-07-06T20:25:40.715105Z","submitted_at":"2025-01-24T17:21:01Z","title":"MedAgentBench: A Realistic Virtual EHR Environment to Benchmark Medical LLM Agents","version":2},"cited_work":{"arxiv_id":"2501.14654","doi":"10.48550/arxiv.2501.14654","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.14654","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Geng, Gloria and Park, Danny and Zou, James and Ng, Andrew Y","venue":"ArXiv.org","work_id":"52ee24a4-e815-4788-8c57-f947197cc09e","year":2025},"citing_paper":{"arxiv_id":"2605.23262","last_updated":"2026-05-22T06:03:01Z","snapshot_observed_at":"2026-08-02T00:22:40.506306Z","submitted_at":"2026-05-22T06:03:01Z","title":"Design and Report Benchmarks for Knowledge Work","version":1},"reference_index":127,"source":"arxiv_source","source_observed_at":"2026-05-25T04:39:14.319133Z"},"links":{"cited_paper":"/paper/2501.14654","citing_paper":"/paper/2605.23262"},"observation_digest":"sha256:750af6096aa6d3c489f184cb547b3574a60602cc9cfb07c7b034d584cffb7d68","observation_id":"a633ad67-e9b3-4c9b-b9af-8addec95764e","resolution":{"observed_at":"2026-05-25T04:40:23.554227Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14654","last_updated":"2025-02-12T05:32:07Z","snapshot_observed_at":"2026-07-06T20:25:40.715105Z","submitted_at":"2025-01-24T17:21:01Z","title":"MedAgentBench: A Realistic Virtual EHR Environment to Benchmark Medical LLM Agents","version":2},"cited_work":{"arxiv_id":"2501.14654","doi":"10.48550/arxiv.2501.14654","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.14654","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Geng, Gloria and Park, Danny and Zou, James and Ng, Andrew Y","venue":"ArXiv.org","work_id":"52ee24a4-e815-4788-8c57-f947197cc09e","year":2025},"citing_paper":{"arxiv_id":"2606.01961","last_updated":"2026-06-03T05:43:27Z","snapshot_observed_at":"2026-07-06T23:42:26.018647Z","submitted_at":"2026-06-01T09:22:55Z","title":"AutoMedBench: Towards Medical AutoResearch with Agentic AI Models","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-06-28T14:38:40.017263Z"},"links":{"cited_paper":"/paper/2501.14654","citing_paper":"/paper/2606.01961"},"observation_digest":"sha256:3592d033156d7a9c4ad75f2cc87fb1f803248477a7eb9b03876f1de17cbc222f","observation_id":"5bb0b44e-83f6-4333-80b3-5cdc13f30f2c","resolution":{"observed_at":"2026-07-01T23:06:21.084048Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14654","last_updated":"2025-02-12T05:32:07Z","snapshot_observed_at":"2026-07-06T20:25:40.715105Z","submitted_at":"2025-01-24T17:21:01Z","title":"MedAgentBench: A Realistic Virtual EHR Environment to Benchmark Medical LLM Agents","version":2},"cited_work":{"arxiv_id":"2501.14654","doi":"10.48550/arxiv.2501.14654","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.14654","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Geng, Gloria and Park, Danny and Zou, James and Ng, Andrew Y","venue":"ArXiv.org","work_id":"52ee24a4-e815-4788-8c57-f947197cc09e","year":2025},"citing_paper":{"arxiv_id":"2606.11740","last_updated":"2026-06-10T07:16:27Z","snapshot_observed_at":"2026-08-02T02:17:16.809620Z","submitted_at":"2026-06-10T07:16:27Z","title":"UniReason-Med: A Shared Grounded Reasoning Interface for 2D-to-3D Transfer in Medical VQA","version":1},"reference_index":157,"source":"arxiv_source","source_observed_at":"2026-06-27T10:21:12.782864Z"},"links":{"cited_paper":"/paper/2501.14654","citing_paper":"/paper/2606.11740"},"observation_digest":"sha256:8fce93b1755204db829e9cc13a89df31b120e4d18f5fcc9abc14fd25211d5c65","observation_id":"49c534cf-00ca-4e6f-a63f-6bdc5a13088f","resolution":{"observed_at":"2026-07-03T09:47:59.364770Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14654","last_updated":"2025-02-12T05:32:07Z","snapshot_observed_at":"2026-07-06T20:25:40.715105Z","submitted_at":"2025-01-24T17:21:01Z","title":"MedAgentBench: A Realistic Virtual EHR Environment to Benchmark Medical LLM Agents","version":2},"cited_work":{"arxiv_id":"2501.14654","doi":"10.48550/arxiv.2501.14654","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.14654","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Geng, Gloria and Park, Danny and Zou, James and Ng, Andrew Y","venue":"ArXiv.org","work_id":"52ee24a4-e815-4788-8c57-f947197cc09e","year":2025},"citing_paper":{"arxiv_id":"2606.12191","last_updated":"2026-06-10T15:15:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-06-10T15:15:01Z","title":"Agentic Environment Engineering for Large Language Models: A Survey of Environment Modeling, Synthesis, Evaluation, and Application","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-06-27T09:46:30.702256Z"},"links":{"cited_paper":"/paper/2501.14654","citing_paper":"/paper/2606.12191"},"observation_digest":"sha256:2e7221e38be576c70fce692bd80b65525b3bf73785ca96a9b450a3cb10cd5551","observation_id":"72c61c02-8158-4888-868b-fdcef8f96452","resolution":{"observed_at":"2026-06-27T09:50:48.306082Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14654","last_updated":"2025-02-12T05:32:07Z","snapshot_observed_at":"2026-07-06T20:25:40.715105Z","submitted_at":"2025-01-24T17:21:01Z","title":"MedAgentBench: A Realistic Virtual EHR Environment to Benchmark Medical LLM Agents","version":2},"cited_work":{"arxiv_id":"2501.14654","doi":"10.48550/arxiv.2501.14654","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.14654","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Geng, Gloria and Park, Danny and Zou, James and Ng, Andrew Y","venue":"ArXiv.org","work_id":"52ee24a4-e815-4788-8c57-f947197cc09e","year":2025},"citing_paper":{"arxiv_id":"2606.12291","last_updated":"2026-06-10T16:27:26Z","snapshot_observed_at":"2026-08-02T15:04:58.080617Z","submitted_at":"2026-06-10T16:27:26Z","title":"Measuring Epistemic Resilience of LLMs Under Misleading Medical Context","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-27T09:30:47.726480Z"},"links":{"cited_paper":"/paper/2501.14654","citing_paper":"/paper/2606.12291"},"observation_digest":"sha256:d609236e438b4c0db7d04108326c45174987e70c05864adf644c4c5397b46cf1","observation_id":"1c4768a3-0fc8-408c-9191-6c9e5ae83fe9","resolution":{"observed_at":"2026-06-27T09:40:47.682854Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14654","last_updated":"2025-02-12T05:32:07Z","snapshot_observed_at":"2026-07-06T20:25:40.715105Z","submitted_at":"2025-01-24T17:21:01Z","title":"MedAgentBench: A Realistic Virtual EHR Environment to Benchmark Medical LLM Agents","version":2},"cited_work":{"arxiv_id":"2501.14654","doi":"10.48550/arxiv.2501.14654","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.14654","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Geng, Gloria and Park, Danny and Zou, James and Ng, Andrew Y","venue":"ArXiv.org","work_id":"52ee24a4-e815-4788-8c57-f947197cc09e","year":2025},"citing_paper":{"arxiv_id":"2606.12736","last_updated":"2026-06-10T22:55:30Z","snapshot_observed_at":"2026-08-02T11:05:19.132987Z","submitted_at":"2026-06-10T22:55:30Z","title":"Benchmarking AI Agents for Addressing Scientific Challenges Across Scales","version":1},"reference_index":145,"source":"pdf_text","source_observed_at":"2026-06-27T09:34:09.347912Z"},"links":{"cited_paper":"/paper/2501.14654","citing_paper":"/paper/2606.12736"},"observation_digest":"sha256:b16bb138edfe6b8c904f887b8a5bdd9c760d155153f388a83a2fbc4eec2e7e50","observation_id":"5ba495e1-35b3-4e47-95ff-6e589c2b6bb4","resolution":{"observed_at":"2026-07-03T11:28:04.327347Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14654","last_updated":"2025-02-12T05:32:07Z","snapshot_observed_at":"2026-07-06T20:25:40.715105Z","submitted_at":"2025-01-24T17:21:01Z","title":"MedAgentBench: A Realistic Virtual EHR Environment to Benchmark Medical LLM Agents","version":2},"cited_work":{"arxiv_id":"2501.14654","doi":"10.48550/arxiv.2501.14654","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.14654","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Geng, Gloria and Park, Danny and Zou, James and Ng, Andrew Y","venue":"ArXiv.org","work_id":"52ee24a4-e815-4788-8c57-f947197cc09e","year":2025},"citing_paper":{"arxiv_id":"2606.13608","last_updated":"2026-06-11T17:23:54Z","snapshot_observed_at":"2026-08-07T11:12:52.411120Z","submitted_at":"2026-06-11T17:23:54Z","title":"AgentBeats: Agentifying Agent Assessment for Openness, Standardization, and Reproducibility","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-27T06:41:41.799596Z"},"links":{"cited_paper":"/paper/2501.14654","citing_paper":"/paper/2606.13608"},"observation_digest":"sha256:56659346b58b0b69477ba15ef2a191b183f38d3a8146850cccdb50abb4a2d6b7","observation_id":"0a341d0f-793c-4991-b85d-cf1c49841d6e","resolution":{"observed_at":"2026-07-03T15:08:33.136412Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14654","last_updated":"2025-02-12T05:32:07Z","snapshot_observed_at":"2026-07-06T20:25:40.715105Z","submitted_at":"2025-01-24T17:21:01Z","title":"MedAgentBench: A Realistic Virtual EHR Environment to Benchmark Medical LLM Agents","version":2},"cited_work":{"arxiv_id":"2501.14654","doi":"10.48550/arxiv.2501.14654","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.14654","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Geng, Gloria and Park, Danny and Zou, James and Ng, Andrew Y","venue":"ArXiv.org","work_id":"52ee24a4-e815-4788-8c57-f947197cc09e","year":2025},"citing_paper":{"arxiv_id":"2606.16723","last_updated":"2026-06-15T13:50:26Z","snapshot_observed_at":"2026-08-07T03:01:48.166687Z","submitted_at":"2026-06-15T13:50:26Z","title":"AgentFairBench: Do LLM Agents Discriminate When They Act?","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-27T03:53:38.554457Z"},"links":{"cited_paper":"/paper/2501.14654","citing_paper":"/paper/2606.16723"},"observation_digest":"sha256:98748b8a8671f124ab70158a2ab6cb2f001db91d1f8f894d5b2120e6c25251e3","observation_id":"79b19494-493c-41c3-87a6-b2bf24459395","resolution":{"observed_at":"2026-07-03T17:38:44.494148Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14654","last_updated":"2025-02-12T05:32:07Z","snapshot_observed_at":"2026-07-06T20:25:40.715105Z","submitted_at":"2025-01-24T17:21:01Z","title":"MedAgentBench: A Realistic Virtual EHR Environment to Benchmark Medical LLM Agents","version":2},"cited_work":{"arxiv_id":"2501.14654","doi":"10.48550/arxiv.2501.14654","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.14654","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Geng, Gloria and Park, Danny and Zou, James and Ng, Andrew Y","venue":"ArXiv.org","work_id":"52ee24a4-e815-4788-8c57-f947197cc09e","year":2025},"citing_paper":{"arxiv_id":"2606.26346","last_updated":"2026-06-24T19:38:21Z","snapshot_observed_at":"2026-08-01T10:31:00.655586Z","submitted_at":"2026-06-24T19:38:21Z","title":"How Do Tool-Augmented LLM Agents Perform on Real-World Energy Analytics Tasks?","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-26T01:34:10.103638Z"},"links":{"cited_paper":"/paper/2501.14654","citing_paper":"/paper/2606.26346"},"observation_digest":"sha256:b45dcaaf3d812f25db41f2df2d18417259c127aaf8d9d8b04fb093df00f2cf6f","observation_id":"a206d07b-58fc-4504-8d1b-d6873ae52798","resolution":{"observed_at":"2026-07-04T15:29:56.673808Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14654","last_updated":"2025-02-12T05:32:07Z","snapshot_observed_at":"2026-07-06T20:25:40.715105Z","submitted_at":"2025-01-24T17:21:01Z","title":"MedAgentBench: A Realistic Virtual EHR Environment to Benchmark Medical LLM Agents","version":2},"cited_work":{"arxiv_id":"2501.14654","doi":"10.48550/arxiv.2501.14654","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.14654","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Geng, Gloria and Park, Danny and Zou, James and Ng, Andrew Y","venue":"ArXiv.org","work_id":"52ee24a4-e815-4788-8c57-f947197cc09e","year":2025},"citing_paper":{"arxiv_id":"2606.28900","last_updated":"2026-06-27T13:14:16Z","snapshot_observed_at":"2026-07-07T00:02:57.622708Z","submitted_at":"2026-06-27T13:14:16Z","title":"MedEvoEval: Evaluating Continual Evolution of Doctor Agents through Simulated Clinical Episodes","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-30T09:33:50.358159Z"},"links":{"cited_paper":"/paper/2501.14654","citing_paper":"/paper/2606.28900"},"observation_digest":"sha256:c8789688f681b42a221b6f81f43944baba4f2999aa8afda0dd90d7cd9046f309","observation_id":"5035f6d7-57dc-4869-9ca7-57a148805abe","resolution":{"observed_at":"2026-06-30T09:34:34.396230Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14654","last_updated":"2025-02-12T05:32:07Z","snapshot_observed_at":"2026-07-06T20:25:40.715105Z","submitted_at":"2025-01-24T17:21:01Z","title":"MedAgentBench: A Realistic Virtual EHR Environment to Benchmark Medical LLM Agents","version":2},"cited_work":{"arxiv_id":"2501.14654","doi":"10.48550/arxiv.2501.14654","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.14654","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Geng, Gloria and Park, Danny and Zou, James and Ng, Andrew Y","venue":"ArXiv.org","work_id":"52ee24a4-e815-4788-8c57-f947197cc09e","year":2025},"citing_paper":{"arxiv_id":"2606.29602","last_updated":"2026-06-28T20:56:57Z","snapshot_observed_at":"2026-07-07T00:03:31.715960Z","submitted_at":"2026-06-28T20:56:57Z","title":"An Empirical Evaluation of Prompt Injection Vulnerabilities in Large Language Models Across Multilingual and Obfuscated Attack Scenarios","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-30T06:51:43.219551Z"},"links":{"cited_paper":"/paper/2501.14654","citing_paper":"/paper/2606.29602"},"observation_digest":"sha256:9154946f2b55fda2b3787a886ac98b4026a0c884d8a8c933b033483ffd81b34d","observation_id":"6040f6c6-0641-41fd-a0a6-4cd8d4a1bd1e","resolution":{"observed_at":"2026-06-30T06:54:20.381699Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14654","last_updated":"2025-02-12T05:32:07Z","snapshot_observed_at":"2026-07-06T20:25:40.715105Z","submitted_at":"2025-01-24T17:21:01Z","title":"MedAgentBench: A Realistic Virtual EHR Environment to Benchmark Medical LLM Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.14654","snapshot_observed_at":"2026-08-02T02:16:39.860668Z","title":"Black, Gloria Geng, Danny Park, James Zou, Andrew Y","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15314","last_updated":"2026-08-04T04:26:54Z","snapshot_observed_at":"2026-08-07T11:11:33.176059Z","submitted_at":"2026-07-15T22:05:23Z","title":"Cura 1T: Specialized Model for Agentic Healthcare","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-02T02:16:39.860668Z"},"links":{"cited_paper":"/paper/2501.14654","citing_paper":"/paper/2607.15314"},"observation_digest":"sha256:8a5dd9d48c9df868065964adc47e793f57dee1c0f56310c1fffc492ea9ed80eb","observation_id":"b3f8ce0b-193e-4b27-88d2-3383809cc627","resolution":{"observed_at":"2026-08-02T02:16:39.860668Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14654","last_updated":"2025-02-12T05:32:07Z","snapshot_observed_at":"2026-07-06T20:25:40.715105Z","submitted_at":"2025-01-24T17:21:01Z","title":"MedAgentBench: A Realistic Virtual EHR Environment to Benchmark Medical LLM Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.14654","snapshot_observed_at":"2026-08-01T13:50:41.542608Z","title":"Jiang, K","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.18999","last_updated":"2026-07-26T15:35:54Z","snapshot_observed_at":"2026-08-06T16:23:45.944884Z","submitted_at":"2026-07-21T11:32:41Z","title":"MedDDC-Eval: Diagnosis-Decoupled Evaluation of Multi-Turn Medical Consultation Agents","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-01T13:50:41.542608Z"},"links":{"cited_paper":"/paper/2501.14654","citing_paper":"/paper/2607.18999"},"observation_digest":"sha256:5c630306c97432287047142e300467b2d751ace98950139f25ed34c7f143e46d","observation_id":"74d61818-c0ee-4ad4-94cd-98cf5826d025","resolution":{"observed_at":"2026-08-01T13:50:41.542608Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14654","last_updated":"2025-02-12T05:32:07Z","snapshot_observed_at":"2026-07-06T20:25:40.715105Z","submitted_at":"2025-01-24T17:21:01Z","title":"MedAgentBench: A Realistic Virtual EHR Environment to Benchmark Medical LLM Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.14654","snapshot_observed_at":"2026-08-05T13:09:08.491553Z","title":"Black, Gloria Geng, Danny Park, Andrew Y","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.03764","last_updated":"2026-08-04T14:51:56Z","snapshot_observed_at":"2026-08-07T11:25:25.719246Z","submitted_at":"2026-08-04T14:51:56Z","title":"GDPevo: Evaluating Agent Self-Evolution on Real Business Tasks","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-05T13:09:08.491553Z"},"links":{"cited_paper":"/paper/2501.14654","citing_paper":"/paper/2608.03764"},"observation_digest":"sha256:5b7276a97b8539a9217d019f6f4d135c692db2dd9ef588ed8e17896f5c4a32a4","observation_id":"d33b646e-a2ec-41a9-8144-9e8dea90379c","resolution":{"observed_at":"2026-08-05T13:09:08.491553Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2501.14654/citation-record","integrity":"/paper/2501.14654/integrity","json":"/paper/2501.14654/citation-record.json","paper":"/paper/2501.14654"},"outbound":[],"paper":{"arxiv_id":"2501.14654","last_updated":"2025-02-12T05:32:07Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-07-06T20:25:40.715105Z","submitted_at":"2025-01-24T17:21:01Z","title":"MedAgentBench: A Realistic Virtual EHR Environment to Benchmark Medical LLM Agents"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 20 inbound Pith citation observations for arXiv:2501.14654."}