{"as_of":"2026-08-05T09:29:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a199f86b1f62b4ed975aa3fec343b4a025418b472b2875e89aa36706950344ad","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":26,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":26,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":26,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":26,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T11:34:03.183686Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":9,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.23802","last_updated":"2025-06-02T04:19:10Z","snapshot_observed_at":"2026-08-01T16:14:33.822709Z","submitted_at":"2025-05-26T22:55:49Z","title":"MedHELM: Holistic Evaluation of Large Language Models for Medical Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.23802","snapshot_observed_at":"2026-08-04T11:34:03.183686Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2510.04033","last_updated":"2026-06-22T22:02:33Z","snapshot_observed_at":"2026-08-04T11:33:58.024115Z","submitted_at":"2025-10-05T04:52:26Z","title":"A global log for medical AI","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-04T11:34:03.183686Z"},"links":{"cited_paper":"/paper/2505.23802","citing_paper":"/paper/2510.04033"},"observation_digest":"sha256:b965fa8502a52091c289f617ae86ef90ac4343ebaba83cc0268c27b11ad5bc22","observation_id":"11e99d4d-feee-4287-ac54-ff0cd2a5ef8f","resolution":{"observed_at":"2026-08-04T11:34:03.183686Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23802","last_updated":"2025-06-02T04:19:10Z","snapshot_observed_at":"2026-08-01T16:14:33.822709Z","submitted_at":"2025-05-26T22:55:49Z","title":"MedHELM: Holistic Evaluation of Large Language Models for Medical Tasks","version":2},"cited_work":{"arxiv_id":"2505.23802","doi":"10.48550/arxiv.2505.23802","metadata_source":"pith","pith_arxiv_id":"2505.23802","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Medhelm: Holistic evaluation of large language models for medical tasks.arXiv preprint arXiv:2505.23802","venue":"cs.CL","work_id":"10fde960-6858-49b3-a93a-c1429deb1170","year":2025},"citing_paper":{"arxiv_id":"2512.19691","last_updated":"2026-04-13T08:00:46Z","snapshot_observed_at":"2026-07-31T18:31:36.321575Z","submitted_at":"2025-12-22T18:59:34Z","title":"Scalable Stewardship of an LLM-Assisted Clinical Benchmark with Physician Oversight","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-16T20:21:40.867354Z"},"links":{"cited_paper":"/paper/2505.23802","citing_paper":"/paper/2512.19691"},"observation_digest":"sha256:8f83f03d919770f963623c2e75b84d4b1233f52414bac0e4dd0c6042fbae3e4a","observation_id":"cc95d819-ab69-4289-91ed-443b94ba985f","resolution":{"observed_at":"2026-05-16T20:23:23.364363Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23802","last_updated":"2025-06-02T04:19:10Z","snapshot_observed_at":"2026-08-01T16:14:33.822709Z","submitted_at":"2025-05-26T22:55:49Z","title":"MedHELM: Holistic Evaluation of Large Language Models for Medical Tasks","version":2},"cited_work":{"arxiv_id":"2505.23802","doi":"10.48550/arxiv.2505.23802","metadata_source":"pith","pith_arxiv_id":"2505.23802","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Medhelm: Holistic evaluation of large language models for medical tasks.arXiv preprint arXiv:2505.23802","venue":"cs.CL","work_id":"10fde960-6858-49b3-a93a-c1429deb1170","year":2025},"citing_paper":{"arxiv_id":"2512.20983","last_updated":"2026-04-07T04:41:41Z","snapshot_observed_at":"2026-08-04T14:29:40.044760Z","submitted_at":"2025-12-24T06:17:21Z","title":"Automatic Replication of LLM Mistakes in Medical Conversations","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-16T20:25:24.722562Z"},"links":{"cited_paper":"/paper/2505.23802","citing_paper":"/paper/2512.20983"},"observation_digest":"sha256:619d9af8b0c6680182c99604ac3401753dcee0a5d714648876fcab3e407300d1","observation_id":"fddf0ff6-8b95-4fdb-ae1e-8ecceac2429b","resolution":{"observed_at":"2026-05-16T20:28:24.177713Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23802","last_updated":"2025-06-02T04:19:10Z","snapshot_observed_at":"2026-08-01T16:14:33.822709Z","submitted_at":"2025-05-26T22:55:49Z","title":"MedHELM: Holistic Evaluation of Large Language Models for Medical Tasks","version":2},"cited_work":{"arxiv_id":"2505.23802","doi":"10.48550/arxiv.2505.23802","metadata_source":"pith","pith_arxiv_id":"2505.23802","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Medhelm: Holistic evaluation of large language models for medical tasks.arXiv preprint arXiv:2505.23802","venue":"cs.CL","work_id":"10fde960-6858-49b3-a93a-c1429deb1170","year":2025},"citing_paper":{"arxiv_id":"2604.02359","last_updated":"2026-03-20T04:31:03Z","snapshot_observed_at":"2026-07-31T07:47:28.480497Z","submitted_at":"2026-03-20T04:31:03Z","title":"Using LLM-as-a-Judge/Jury to Advance Scalable, Clinically-Validated Safety Evaluations of Model Responses to Users Demonstrating Psychosis","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-15T09:06:33.531027Z"},"links":{"cited_paper":"/paper/2505.23802","citing_paper":"/paper/2604.02359"},"observation_digest":"sha256:81fc65e49428aef7b68168473164852f67db96d9265fdbe033de25559f9fb084","observation_id":"efeacace-d7dc-428f-b398-55a72c9469c7","resolution":{"observed_at":"2026-05-15T09:09:53.374917Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23802","last_updated":"2025-06-02T04:19:10Z","snapshot_observed_at":"2026-08-01T16:14:33.822709Z","submitted_at":"2025-05-26T22:55:49Z","title":"MedHELM: Holistic Evaluation of Large Language Models for Medical Tasks","version":2},"cited_work":{"arxiv_id":"2505.23802","doi":"10.48550/arxiv.2505.23802","metadata_source":"pith","pith_arxiv_id":"2505.23802","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Medhelm: Holistic evaluation of large language models for medical tasks.arXiv preprint arXiv:2505.23802","venue":"cs.CL","work_id":"10fde960-6858-49b3-a93a-c1429deb1170","year":2025},"citing_paper":{"arxiv_id":"2604.11133","last_updated":"2026-04-15T02:31:20Z","snapshot_observed_at":"2026-08-01T02:09:41.306280Z","submitted_at":"2026-04-13T07:44:41Z","title":"How Robust Are Large Language Models for Clinical Numeracy? An Empirical Study on Numerical Reasoning Abilities in Clinical Contexts","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T15:41:21.704608Z"},"links":{"cited_paper":"/paper/2505.23802","citing_paper":"/paper/2604.11133"},"observation_digest":"sha256:0b0520398029a7e30bbf026376878874a9d44cea8358cb9cee6e7d41f7477d3f","observation_id":"176be442-07bd-44c1-a6a2-1d62c26af238","resolution":{"observed_at":"2026-05-11T10:01:03.373551Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23802","last_updated":"2025-06-02T04:19:10Z","snapshot_observed_at":"2026-08-01T16:14:33.822709Z","submitted_at":"2025-05-26T22:55:49Z","title":"MedHELM: Holistic Evaluation of Large Language Models for Medical Tasks","version":2},"cited_work":{"arxiv_id":"2505.23802","doi":"10.48550/arxiv.2505.23802","metadata_source":"pith","pith_arxiv_id":"2505.23802","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Medhelm: Holistic evaluation of large language models for medical tasks.arXiv preprint arXiv:2505.23802","venue":"cs.CL","work_id":"10fde960-6858-49b3-a93a-c1429deb1170","year":2025},"citing_paper":{"arxiv_id":"2604.24700","last_updated":"2026-04-27T17:04:17Z","snapshot_observed_at":"2026-08-04T04:55:30.006592Z","submitted_at":"2026-04-27T17:04:17Z","title":"Green Shielding: A User-Centric Approach Towards Trustworthy AI","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-08T03:43:54.896449Z"},"links":{"cited_paper":"/paper/2505.23802","citing_paper":"/paper/2604.24700"},"observation_digest":"sha256:329f0e02114290f939daccf2f957a1e907e2e7b0b210bb6f037489a45df4f281","observation_id":"327c5508-4d08-437b-88e3-682553db506d","resolution":{"observed_at":"2026-05-11T21:56:24.527571Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23802","last_updated":"2025-06-02T04:19:10Z","snapshot_observed_at":"2026-08-01T16:14:33.822709Z","submitted_at":"2025-05-26T22:55:49Z","title":"MedHELM: Holistic Evaluation of Large Language Models for Medical Tasks","version":2},"cited_work":{"arxiv_id":"2505.23802","doi":"10.48550/arxiv.2505.23802","metadata_source":"pith","pith_arxiv_id":"2505.23802","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Medhelm: Holistic evaluation of large language models for medical tasks.arXiv preprint arXiv:2505.23802","venue":"cs.CL","work_id":"10fde960-6858-49b3-a93a-c1429deb1170","year":2025},"citing_paper":{"arxiv_id":"2605.04012","last_updated":"2026-05-10T21:01:37Z","snapshot_observed_at":"2026-08-02T05:52:52.199459Z","submitted_at":"2026-05-05T17:36:12Z","title":"SymptomAI: Toward a Conversational AI Agent for Everyday Symptom Assessment","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-07T16:17:39.923337Z"},"links":{"cited_paper":"/paper/2505.23802","citing_paper":"/paper/2605.04012"},"observation_digest":"sha256:5e7a616dea53b65e53ff2c71612adf058ea0400940529dbcb102515d77bb3262","observation_id":"6e0e4844-4077-49ab-a851-21f86558be09","resolution":{"observed_at":"2026-05-11T23:46:53.503600Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23802","last_updated":"2025-06-02T04:19:10Z","snapshot_observed_at":"2026-08-01T16:14:33.822709Z","submitted_at":"2025-05-26T22:55:49Z","title":"MedHELM: Holistic Evaluation of Large Language Models for Medical Tasks","version":2},"cited_work":{"arxiv_id":"2505.23802","doi":"10.48550/arxiv.2505.23802","metadata_source":"pith","pith_arxiv_id":"2505.23802","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Medhelm: Holistic evaluation of large language models for medical tasks.arXiv preprint arXiv:2505.23802","venue":"cs.CL","work_id":"10fde960-6858-49b3-a93a-c1429deb1170","year":2025},"citing_paper":{"arxiv_id":"2605.04012","last_updated":"2026-05-10T21:01:37Z","snapshot_observed_at":"2026-08-02T05:52:52.199459Z","submitted_at":"2026-05-05T17:36:12Z","title":"SymptomAI: Toward a Conversational AI Agent for Everyday Symptom Assessment","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-12T02:21:16.825681Z"},"links":{"cited_paper":"/paper/2505.23802","citing_paper":"/paper/2605.04012"},"observation_digest":"sha256:4addea086119f7b29cdc2800ca4efe6ef20009430e8e161c8acf39e9855fa776","observation_id":"d561dc02-1a1b-4291-9152-86bdca298f24","resolution":{"observed_at":"2026-05-12T07:41:42.864862Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23802","last_updated":"2025-06-02T04:19:10Z","snapshot_observed_at":"2026-08-01T16:14:33.822709Z","submitted_at":"2025-05-26T22:55:49Z","title":"MedHELM: Holistic Evaluation of Large Language Models for Medical Tasks","version":2},"cited_work":{"arxiv_id":"2505.23802","doi":"10.48550/arxiv.2505.23802","metadata_source":"pith","pith_arxiv_id":"2505.23802","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Medhelm: Holistic evaluation of large language models for medical tasks.arXiv preprint arXiv:2505.23802","venue":"cs.CL","work_id":"10fde960-6858-49b3-a93a-c1429deb1170","year":2025},"citing_paper":{"arxiv_id":"2605.08445","last_updated":"2026-05-08T20:12:14Z","snapshot_observed_at":"2026-08-02T06:12:10.270159Z","submitted_at":"2026-05-08T20:12:14Z","title":"Measuring What Matters: Benchmarking Generative, Multimodal, and Agentic AI in Healthcare","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-12T01:26:28.419570Z"},"links":{"cited_paper":"/paper/2505.23802","citing_paper":"/paper/2605.08445"},"observation_digest":"sha256:c21eb52ade00fabb29ae1ab0331bd611bfc147d9797a7b2e3692f9277bb2ce07","observation_id":"1cb41f50-496a-47e3-abb4-b9fbe9bf493a","resolution":{"observed_at":"2026-05-12T07:56:36.287260Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23802","last_updated":"2025-06-02T04:19:10Z","snapshot_observed_at":"2026-08-01T16:14:33.822709Z","submitted_at":"2025-05-26T22:55:49Z","title":"MedHELM: Holistic Evaluation of Large Language Models for Medical Tasks","version":2},"cited_work":{"arxiv_id":"2505.23802","doi":"10.48550/arxiv.2505.23802","metadata_source":"pith","pith_arxiv_id":"2505.23802","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Medhelm: Holistic evaluation of large language models for medical tasks.arXiv preprint arXiv:2505.23802","venue":"cs.CL","work_id":"10fde960-6858-49b3-a93a-c1429deb1170","year":2025},"citing_paper":{"arxiv_id":"2605.08685","last_updated":"2026-05-09T04:49:56Z","snapshot_observed_at":"2026-08-02T14:44:33.851541Z","submitted_at":"2026-05-09T04:49:56Z","title":"Event Fields: Learning Latent Event Structure for Waveform Foundation Models","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-12T01:16:48.039349Z"},"links":{"cited_paper":"/paper/2505.23802","citing_paper":"/paper/2605.08685"},"observation_digest":"sha256:275c2e096881cc2d6608f72039e64facbe40a7221f7b5ba5407cbd3afdfb8787","observation_id":"e8838776-c841-4e93-b451-6ca2540bee98","resolution":{"observed_at":"2026-05-12T08:06:31.984899Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23802","last_updated":"2025-06-02T04:19:10Z","snapshot_observed_at":"2026-08-01T16:14:33.822709Z","submitted_at":"2025-05-26T22:55:49Z","title":"MedHELM: Holistic Evaluation of Large Language Models for Medical Tasks","version":2},"cited_work":{"arxiv_id":"2505.23802","doi":"10.48550/arxiv.2505.23802","metadata_source":"pith","pith_arxiv_id":"2505.23802","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Medhelm: Holistic evaluation of large language models for medical tasks.arXiv preprint arXiv:2505.23802","venue":"cs.CL","work_id":"10fde960-6858-49b3-a93a-c1429deb1170","year":2025},"citing_paper":{"arxiv_id":"2605.09584","last_updated":"2026-05-10T14:51:31Z","snapshot_observed_at":"2026-07-06T23:21:39.966502Z","submitted_at":"2026-05-10T14:51:31Z","title":"CLR-voyance: Reinforcing Open-Ended Reasoning for Inpatient Clinical Decision Support with Outcome-Aware Rubrics","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-05-12T04:32:16.930291Z"},"links":{"cited_paper":"/paper/2505.23802","citing_paper":"/paper/2605.09584"},"observation_digest":"sha256:085df1f5fe43ea718e65c9df1f58c44847a437ab5ee42427584f29550d003c15","observation_id":"aad9683c-c57b-4bdb-b23b-a2ceec53f2c0","resolution":{"observed_at":"2026-05-12T06:11:23.506120Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23802","last_updated":"2025-06-02T04:19:10Z","snapshot_observed_at":"2026-08-01T16:14:33.822709Z","submitted_at":"2025-05-26T22:55:49Z","title":"MedHELM: Holistic Evaluation of Large Language Models for Medical Tasks","version":2},"cited_work":{"arxiv_id":"2505.23802","doi":"10.48550/arxiv.2505.23802","metadata_source":"pith","pith_arxiv_id":"2505.23802","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Medhelm: Holistic evaluation of large language models for medical tasks.arXiv preprint arXiv:2505.23802","venue":"cs.CL","work_id":"10fde960-6858-49b3-a93a-c1429deb1170","year":2025},"citing_paper":{"arxiv_id":"2605.09675","last_updated":"2026-05-10T17:45:01Z","snapshot_observed_at":"2026-07-06T23:21:44.900680Z","submitted_at":"2026-05-10T17:45:01Z","title":"CodeClinic: Evaluating Automation of Coding Skills for Clinical Reasoning Agents","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-05-12T02:17:41.650948Z"},"links":{"cited_paper":"/paper/2505.23802","citing_paper":"/paper/2605.09675"},"observation_digest":"sha256:e0838fb0ec6a4bb2e53c1deb192f5ebf054feba856180d508e00a7cd1acbec95","observation_id":"5a1ce8a6-9fc1-44fc-a5ce-6d63febeb983","resolution":{"observed_at":"2026-05-12T07:41:46.226036Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23802","last_updated":"2025-06-02T04:19:10Z","snapshot_observed_at":"2026-08-01T16:14:33.822709Z","submitted_at":"2025-05-26T22:55:49Z","title":"MedHELM: Holistic Evaluation of Large Language Models for Medical Tasks","version":2},"cited_work":{"arxiv_id":"2505.23802","doi":"10.48550/arxiv.2505.23802","metadata_source":"pith","pith_arxiv_id":"2505.23802","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Medhelm: Holistic evaluation of large language models for medical tasks.arXiv preprint arXiv:2505.23802","venue":"cs.CL","work_id":"10fde960-6858-49b3-a93a-c1429deb1170","year":2025},"citing_paper":{"arxiv_id":"2605.09765","last_updated":"2026-05-10T21:25:41Z","snapshot_observed_at":"2026-07-06T23:21:49.499951Z","submitted_at":"2026-05-10T21:25:41Z","title":"WISTERIA: Learning Clinical Representations from Noisy Supervision via Multi-View Consistency in Electronic Health Records","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-12T02:17:54.498339Z"},"links":{"cited_paper":"/paper/2505.23802","citing_paper":"/paper/2605.09765"},"observation_digest":"sha256:139457062643817f5152ff8856f3485188c0c016048c9b8ae13ecc95d2a25098","observation_id":"c6de4e66-d417-4eeb-aa60-cd40fb5d06c2","resolution":{"observed_at":"2026-05-12T07:41:45.088746Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23802","last_updated":"2025-06-02T04:19:10Z","snapshot_observed_at":"2026-08-01T16:14:33.822709Z","submitted_at":"2025-05-26T22:55:49Z","title":"MedHELM: Holistic Evaluation of Large Language Models for Medical Tasks","version":2},"cited_work":{"arxiv_id":"2505.23802","doi":"10.48550/arxiv.2505.23802","metadata_source":"pith","pith_arxiv_id":"2505.23802","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Medhelm: Holistic evaluation of large language models for medical tasks.arXiv preprint arXiv:2505.23802","venue":"cs.CL","work_id":"10fde960-6858-49b3-a93a-c1429deb1170","year":2025},"citing_paper":{"arxiv_id":"2605.11206","last_updated":"2026-05-13T08:57:14Z","snapshot_observed_at":"2026-07-06T23:23:06.755816Z","submitted_at":"2026-05-11T20:21:04Z","title":"Instructions Shape Production of Language, not Processing","version":1},"reference_index":121,"source":"arxiv_source","source_observed_at":"2026-05-13T03:09:02.902912Z"},"links":{"cited_paper":"/paper/2505.23802","citing_paper":"/paper/2605.11206"},"observation_digest":"sha256:f68e93b4513f68f736aa24bc3d67bc398abd2ca95a2e00ea01163f7f83f72ba8","observation_id":"fa76b768-a8be-47a8-b2fb-57d57c8c3083","resolution":{"observed_at":"2026-05-13T03:12:09.414953Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23802","last_updated":"2025-06-02T04:19:10Z","snapshot_observed_at":"2026-08-01T16:14:33.822709Z","submitted_at":"2025-05-26T22:55:49Z","title":"MedHELM: Holistic Evaluation of Large Language Models for Medical Tasks","version":2},"cited_work":{"arxiv_id":"2505.23802","doi":"10.48550/arxiv.2505.23802","metadata_source":"pith","pith_arxiv_id":"2505.23802","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Medhelm: Holistic evaluation of large language models for medical tasks.arXiv preprint arXiv:2505.23802","venue":"cs.CL","work_id":"10fde960-6858-49b3-a93a-c1429deb1170","year":2025},"citing_paper":{"arxiv_id":"2605.11206","last_updated":"2026-05-13T08:57:14Z","snapshot_observed_at":"2026-07-06T23:23:06.755816Z","submitted_at":"2026-05-11T20:21:04Z","title":"Instructions Shape Production of Language, not Processing","version":2},"reference_index":121,"source":"arxiv_source","source_observed_at":"2026-05-14T21:02:02.135970Z"},"links":{"cited_paper":"/paper/2505.23802","citing_paper":"/paper/2605.11206"},"observation_digest":"sha256:d64462e613046ba5bd2006576751af2766aeb2ec8685b4b4d2f391e91710f981","observation_id":"d81f6e33-e0ee-4a38-82fb-3d17b56c1898","resolution":{"observed_at":"2026-05-14T21:02:58.194611Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23802","last_updated":"2025-06-02T04:19:10Z","snapshot_observed_at":"2026-08-01T16:14:33.822709Z","submitted_at":"2025-05-26T22:55:49Z","title":"MedHELM: Holistic Evaluation of Large Language Models for Medical Tasks","version":2},"cited_work":{"arxiv_id":"2505.23802","doi":"10.48550/arxiv.2505.23802","metadata_source":"pith","pith_arxiv_id":"2505.23802","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Medhelm: Holistic evaluation of large language models for medical tasks.arXiv preprint arXiv:2505.23802","venue":"cs.CL","work_id":"10fde960-6858-49b3-a93a-c1429deb1170","year":2025},"citing_paper":{"arxiv_id":"2605.16679","last_updated":"2026-05-19T05:51:20Z","snapshot_observed_at":"2026-08-02T09:51:22.510077Z","submitted_at":"2026-05-15T22:34:31Z","title":"CHI-Bench: Can AI Agents Automate End-to-End, Long-Horizon, Policy-Rich Healthcare Workflows?","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-20T17:45:02.896703Z"},"links":{"cited_paper":"/paper/2505.23802","citing_paper":"/paper/2605.16679"},"observation_digest":"sha256:6961650fc847c383ac43b09780060a0553e9d9dca98a5a18fc45c219d71df32c","observation_id":"744597a7-4cad-4739-9e57-cd78a33353af","resolution":{"observed_at":"2026-05-20T17:48:48.910341Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23802","last_updated":"2025-06-02T04:19:10Z","snapshot_observed_at":"2026-08-01T16:14:33.822709Z","submitted_at":"2025-05-26T22:55:49Z","title":"MedHELM: Holistic Evaluation of Large Language Models for Medical Tasks","version":2},"cited_work":{"arxiv_id":"2505.23802","doi":"10.48550/arxiv.2505.23802","metadata_source":"pith","pith_arxiv_id":"2505.23802","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Medhelm: Holistic evaluation of large language models for medical tasks.arXiv preprint arXiv:2505.23802","venue":"cs.CL","work_id":"10fde960-6858-49b3-a93a-c1429deb1170","year":2025},"citing_paper":{"arxiv_id":"2605.17765","last_updated":"2026-05-18T02:32:50Z","snapshot_observed_at":"2026-08-02T07:46:25.913418Z","submitted_at":"2026-05-18T02:32:50Z","title":"AURORA: Contextual Orthogonalization for Geometric Representation Learning in Healthcare Foundation Models","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-20T12:06:33.875494Z"},"links":{"cited_paper":"/paper/2505.23802","citing_paper":"/paper/2605.17765"},"observation_digest":"sha256:cc5401a5b4bd4340554fa59c46f8f47cdfdb0dbec2261103d7450cc02040569c","observation_id":"191a7621-23b8-420d-ac62-46b885acc01f","resolution":{"observed_at":"2026-05-20T12:08:15.521584Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23802","last_updated":"2025-06-02T04:19:10Z","snapshot_observed_at":"2026-08-01T16:14:33.822709Z","submitted_at":"2025-05-26T22:55:49Z","title":"MedHELM: Holistic Evaluation of Large Language Models for Medical Tasks","version":2},"cited_work":{"arxiv_id":"2505.23802","doi":"10.48550/arxiv.2505.23802","metadata_source":"pith","pith_arxiv_id":"2505.23802","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Medhelm: Holistic evaluation of large language models for medical tasks.arXiv preprint arXiv:2505.23802","venue":"cs.CL","work_id":"10fde960-6858-49b3-a93a-c1429deb1170","year":2025},"citing_paper":{"arxiv_id":"2606.01961","last_updated":"2026-06-03T05:43:27Z","snapshot_observed_at":"2026-07-06T23:42:26.018647Z","submitted_at":"2026-06-01T09:22:55Z","title":"AutoMedBench: Towards Medical AutoResearch with Agentic AI Models","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-28T14:38:40.017263Z"},"links":{"cited_paper":"/paper/2505.23802","citing_paper":"/paper/2606.01961"},"observation_digest":"sha256:5f71bc04b5e3db490e8145cdf009ccd2eb328c16d8900a3ca88e21171eb04361","observation_id":"09695644-9950-4a37-a589-a75502a21ca0","resolution":{"observed_at":"2026-07-01T23:06:21.116582Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23802","last_updated":"2025-06-02T04:19:10Z","snapshot_observed_at":"2026-08-01T16:14:33.822709Z","submitted_at":"2025-05-26T22:55:49Z","title":"MedHELM: Holistic Evaluation of Large Language Models for Medical Tasks","version":2},"cited_work":{"arxiv_id":"2505.23802","doi":"10.48550/arxiv.2505.23802","metadata_source":"pith","pith_arxiv_id":"2505.23802","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Medhelm: Holistic evaluation of large language models for medical tasks.arXiv preprint arXiv:2505.23802","venue":"cs.CL","work_id":"10fde960-6858-49b3-a93a-c1429deb1170","year":2025},"citing_paper":{"arxiv_id":"2606.07853","last_updated":"2026-06-05T21:29:39Z","snapshot_observed_at":"2026-08-02T04:31:19.934605Z","submitted_at":"2026-06-05T21:29:39Z","title":"Beyond English benchmarks: clinical llm evaluation in Brazilian Portuguese","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-27T21:40:16.052239Z"},"links":{"cited_paper":"/paper/2505.23802","citing_paper":"/paper/2606.07853"},"observation_digest":"sha256:34bbcf509537e18160839c90829ab3aa51e8f8c31fef233f442be43e6bf4443e","observation_id":"62754d03-7ef2-4445-b55c-ad591e931ff6","resolution":{"observed_at":"2026-07-02T19:07:17.939001Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23802","last_updated":"2025-06-02T04:19:10Z","snapshot_observed_at":"2026-08-01T16:14:33.822709Z","submitted_at":"2025-05-26T22:55:49Z","title":"MedHELM: Holistic Evaluation of Large Language Models for Medical Tasks","version":2},"cited_work":{"arxiv_id":"2505.23802","doi":"10.48550/arxiv.2505.23802","metadata_source":"pith","pith_arxiv_id":"2505.23802","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Medhelm: Holistic evaluation of large language models for medical tasks.arXiv preprint arXiv:2505.23802","venue":"cs.CL","work_id":"10fde960-6858-49b3-a93a-c1429deb1170","year":2025},"citing_paper":{"arxiv_id":"2606.31608","last_updated":"2026-06-30T12:56:42Z","snapshot_observed_at":"2026-07-07T00:05:17.259491Z","submitted_at":"2026-06-30T12:56:42Z","title":"CLExEval: A Human-in-the-Loop Framework for Qualitative Evaluation of LLM Clinical Reasoning","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-07-01T05:33:52.027771Z"},"links":{"cited_paper":"/paper/2505.23802","citing_paper":"/paper/2606.31608"},"observation_digest":"sha256:68116c98b8228dd7f135e1b70c08b0b0ffaf825a61d36a82ed26cb854bd9d6bf","observation_id":"26c2190a-afd8-435c-a37b-521b4d7a93bf","resolution":{"observed_at":"2026-07-01T10:25:41.545608Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23802","last_updated":"2025-06-02T04:19:10Z","snapshot_observed_at":"2026-08-01T16:14:33.822709Z","submitted_at":"2025-05-26T22:55:49Z","title":"MedHELM: Holistic Evaluation of Large Language Models for Medical Tasks","version":2},"cited_work":{"arxiv_id":"2505.23802","doi":"10.48550/arxiv.2505.23802","metadata_source":"pith","pith_arxiv_id":"2505.23802","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Medhelm: Holistic evaluation of large language models for medical tasks.arXiv preprint arXiv:2505.23802","venue":"cs.CL","work_id":"10fde960-6858-49b3-a93a-c1429deb1170","year":2025},"citing_paper":{"arxiv_id":"2607.02175","last_updated":"2026-07-02T13:46:24Z","snapshot_observed_at":"2026-07-07T00:07:37.820438Z","submitted_at":"2026-07-02T13:46:24Z","title":"A rubric-based controlled comparison of frontier language models on expert-authored clinical reasoning tasks","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-07-03T14:05:58.062268Z"},"links":{"cited_paper":"/paper/2505.23802","citing_paper":"/paper/2607.02175"},"observation_digest":"sha256:92c089be10d89f741ed191671a45ff9de038b2ff887c2cb36b7f4c31af729e42","observation_id":"d830393c-29e0-477b-af91-f4f255af879b","resolution":{"observed_at":"2026-07-03T14:08:21.387402Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23802","last_updated":"2025-06-02T04:19:10Z","snapshot_observed_at":"2026-08-01T16:14:33.822709Z","submitted_at":"2025-05-26T22:55:49Z","title":"MedHELM: Holistic Evaluation of Large Language Models for Medical Tasks","version":2},"cited_work":{"arxiv_id":"2505.23802","doi":"10.48550/arxiv.2505.23802","metadata_source":"pith","pith_arxiv_id":"2505.23802","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Medhelm: Holistic evaluation of large language models for medical tasks.arXiv preprint arXiv:2505.23802","venue":"cs.CL","work_id":"10fde960-6858-49b3-a93a-c1429deb1170","year":2025},"citing_paper":{"arxiv_id":"2607.07761","last_updated":"2026-07-08T15:19:37Z","snapshot_observed_at":"2026-08-01T15:00:20.669918Z","submitted_at":"2026-07-08T15:19:37Z","title":"Aligning Clinical Needs and AI Capabilities: A Survey on LLMs for Medical Reasoning","version":1},"reference_index":156,"source":"pdf_text","source_observed_at":"2026-07-10T18:50:22.827472Z"},"links":{"cited_paper":"/paper/2505.23802","citing_paper":"/paper/2607.07761"},"observation_digest":"sha256:f352b49bc2d3f9e894de18898900884a5dbd96fd95e1b5e56351f7d01447747d","observation_id":"4975a1b2-199d-4813-9899-ab15e37aa937","resolution":{"observed_at":"2026-07-10T18:57:31.509556Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23802","last_updated":"2025-06-02T04:19:10Z","snapshot_observed_at":"2026-08-01T16:14:33.822709Z","submitted_at":"2025-05-26T22:55:49Z","title":"MedHELM: Holistic Evaluation of Large Language Models for Medical Tasks","version":2},"cited_work":{"arxiv_id":"2505.23802","doi":"10.48550/arxiv.2505.23802","metadata_source":"pith","pith_arxiv_id":"2505.23802","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Medhelm: Holistic evaluation of large language models for medical tasks.arXiv preprint arXiv:2505.23802","venue":"cs.CL","work_id":"10fde960-6858-49b3-a93a-c1429deb1170","year":2025},"citing_paper":{"arxiv_id":"2607.08257","last_updated":"2026-07-09T09:03:42Z","snapshot_observed_at":"2026-08-02T23:36:40.452871Z","submitted_at":"2026-07-09T09:03:42Z","title":"MentalHospital: A Virtual Environment for Evaluating Psychiatric Clinical Encounters","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-07-10T10:30:27.256710Z"},"links":{"cited_paper":"/paper/2505.23802","citing_paper":"/paper/2607.08257"},"observation_digest":"sha256:6da8cd689eab63816b36723835136c8183b79a66705e2ff1fd646e90b43304fc","observation_id":"c54336c8-775b-4811-a61c-ccb61518a1c0","resolution":{"observed_at":"2026-07-10T10:37:01.669041Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23802","last_updated":"2025-06-02T04:19:10Z","snapshot_observed_at":"2026-08-01T16:14:33.822709Z","submitted_at":"2025-05-26T22:55:49Z","title":"MedHELM: Holistic Evaluation of Large Language Models for Medical Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.23802","snapshot_observed_at":"2026-07-14T06:30:16.612345Z","title":"Medhelm: Holistic evaluation of large language models for medical tasks.arXiv preprint arXiv:2505.23802, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.11175","last_updated":"2026-07-13T07:16:50Z","snapshot_observed_at":"2026-07-16T23:19:14.481825Z","submitted_at":"2026-07-13T07:16:50Z","title":"The Path to Self-Evolving Clinical Systems: Scaling Medical Agents from Assistance to Autonomy","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-07-14T06:30:16.612345Z"},"links":{"cited_paper":"/paper/2505.23802","citing_paper":"/paper/2607.11175"},"observation_digest":"sha256:6f9b1eb012d7fad2fe8878b878edbdf97476a65a50b7f914e50ec03850a0787b","observation_id":"da873bf5-463f-41dd-8348-386ac3616068","resolution":{"observed_at":"2026-07-14T06:30:16.612345Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23802","last_updated":"2025-06-02T04:19:10Z","snapshot_observed_at":"2026-08-01T16:14:33.822709Z","submitted_at":"2025-05-26T22:55:49Z","title":"MedHELM: Holistic Evaluation of Large Language Models for Medical Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.23802","snapshot_observed_at":"2026-08-01T13:50:41.494173Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.18999","last_updated":"2026-07-26T15:35:54Z","snapshot_observed_at":"2026-08-01T13:50:38.848506Z","submitted_at":"2026-07-21T11:32:41Z","title":"MedDDC-Eval: Diagnosis-Decoupled Evaluation of Multi-Turn Medical Consultation Agents","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-01T13:50:41.494173Z"},"links":{"cited_paper":"/paper/2505.23802","citing_paper":"/paper/2607.18999"},"observation_digest":"sha256:5dab46847d02b5e41a996bb136e59db5c6012ddc264a92906043dc60c4945aa4","observation_id":"b7b69def-859d-41e1-bfc4-b7e23a9c5660","resolution":{"observed_at":"2026-08-01T13:50:41.494173Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23802","last_updated":"2025-06-02T04:19:10Z","snapshot_observed_at":"2026-08-01T16:14:33.822709Z","submitted_at":"2025-05-26T22:55:49Z","title":"MedHELM: Holistic Evaluation of Large Language Models for Medical Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.23802","snapshot_observed_at":"2026-08-04T05:39:50.996391Z","title":"2025 , eprint =","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.02520","last_updated":"2026-08-03T17:17:29Z","snapshot_observed_at":"2026-08-05T09:24:56.456885Z","submitted_at":"2026-08-03T17:17:29Z","title":"MedPRESS: A Multi-turn Benchmark for Patient-Pressure-Induced Medical Sycophancy in LLMs","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-04T05:39:50.996391Z"},"links":{"cited_paper":"/paper/2505.23802","citing_paper":"/paper/2608.02520"},"observation_digest":"sha256:98d92913cfbda0ee61385289750f901b911c942921b06dd1e8cbd11819af8c92","observation_id":"98e3ffa5-fa46-45a7-86ad-a53a6fa0da2a","resolution":{"observed_at":"2026-08-04T05:39:50.996391Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2505.23802/citation-record","integrity":"/paper/2505.23802/integrity","json":"/paper/2505.23802/citation-record.json","paper":"/paper/2505.23802"},"outbound":[],"paper":{"arxiv_id":"2505.23802","last_updated":"2025-06-02T04:19:10Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-01T16:14:33.822709Z","submitted_at":"2025-05-26T22:55:49Z","title":"MedHELM: Holistic Evaluation of Large Language Models for Medical Tasks"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 5 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 26 inbound Pith citation observations for arXiv:2505.23802."}