{"as_of":"2026-08-23T10:00:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:fea50e10dfbfb46e70d3fc709f300446d7e7bc6a83f3bcf130f74fbd5b93bc88","coverage":[{"denominator":30,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":30,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-16T14:26:38.517787Z","state":"measured"},{"denominator":118,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":118,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-23T06:30:58.430688+00:00","state":"measured"},{"denominator":88,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":88,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T05:20:20.277083Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":1,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-16T05:20:20.277083Z","title":"Medxpertqa: Benchmarking expert-level medical reasoning and understanding.arXiv preprint arXiv:2501.18362, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2504.20930","last_updated":"2025-05-21T08:50:11Z","snapshot_observed_at":"2026-08-17T17:43:10.588783Z","submitted_at":"2025-04-29T16:48:23Z","title":"ChestX-Reasoner: Advancing Radiology Foundation Models with Reasoning through Step-by-Step Verification","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-16T05:20:20.277083Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2504.20930"},"observation_digest":"sha256:f543c770d25760b410bab1daa2eb295ab2b0237d8368653d856bcf417f949d3c","observation_id":"dce3a636-1c70-416a-be7a-0d7582ac3b3c","resolution":{"observed_at":"2026-08-16T05:20:20.277083Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-15T22:33:24.554504Z","title":"Medxpertqa: Benchmarking expert-level medical reasoning and understanding.arXiv preprint arXiv:2501.18362, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.07889","last_updated":"2026-07-27T01:08:26Z","snapshot_observed_at":"2026-08-20T01:47:45.036432Z","submitted_at":"2025-05-11T09:42:24Z","title":"BioProBench: A Corpus and Benchmark for Biological Protocol Reasoning in Autonomous Science","version":4},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-15T22:33:24.554504Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2505.07889"},"observation_digest":"sha256:74c59d2bb05f73a5d3ef92de31d49d1726cbfafe9e74e602de7b6631248459ce","observation_id":"00c3fdb2-6f00-44be-bef9-dc76724c0ff6","resolution":{"observed_at":"2026-08-15T22:33:24.554504Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-15T20:57:56.132621Z","title":"I think the answer is {wrong answer}, but I am not sure. Let me think again","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.11462","last_updated":"2025-06-24T03:27:30Z","snapshot_observed_at":"2026-08-22T05:53:02.163448Z","submitted_at":"2025-05-16T17:16:27Z","title":"Disentangling Reasoning and Knowledge in Medical Large Language Models","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-15T20:57:56.132621Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2505.11462"},"observation_digest":"sha256:3b7fd812bbd344a24597944751edfc38f98bd9d8c4296ed4f928ab12e93962e1","observation_id":"a6deb072-9fe2-4cc4-ba96-89e80df68f58","resolution":{"observed_at":"2026-08-15T20:57:56.132621Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-07T15:42:25.004807Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.14107","last_updated":"2026-08-17T08:06:21Z","snapshot_observed_at":"2026-08-20T23:15:26.177310Z","submitted_at":"2025-05-20T09:14:53Z","title":"DiagnosisArena: Benchmarking Diagnostic Reasoning for Large Language Models","version":4},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T15:42:25.004807Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2505.14107"},"observation_digest":"sha256:976cb6d740eff17492e37b055559d1fd8324153022679957099bc090af92430a","observation_id":"ee58effb-e067-41e3-bf66-4c5f270dcafe","resolution":{"observed_at":"2026-08-07T15:42:25.004807Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-07T14:41:38.423359Z","title":"Medxpertqa: Bench- marking expert-level medical reasoning and understanding,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17952","last_updated":"2025-05-23T14:27:37Z","snapshot_observed_at":"2026-08-23T05:50:39.911254Z","submitted_at":"2025-05-23T14:27:37Z","title":"Beyond Distillation: Pushing the Limits of Medical LLM Reasoning with Minimalist Rule-Based RL","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T14:41:38.423359Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2505.17952"},"observation_digest":"sha256:e1739363ae39824553390c03f994d9f7f907ac2450d9b466c54618a8ff555b96","observation_id":"f4b35547-716f-4666-acfb-e821b41c857e","resolution":{"observed_at":"2026-08-07T14:41:38.423359Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-07T14:39:49.370355Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.18283","last_updated":"2025-05-23T18:28:59Z","snapshot_observed_at":"2026-08-20T01:26:29.753560Z","submitted_at":"2025-05-23T18:28:59Z","title":"TAGS: A Test-Time Generalist-Specialist Framework with Retrieval-Augmented Reasoning and Verification","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-07T14:39:49.370355Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2505.18283"},"observation_digest":"sha256:6f1bbdb4acd45246fbb4d9c03bfbd387d0b6c75dc0a4ee895bc7ab6c1ead9e1b","observation_id":"b7b2aba6-1284-48d0-ae5a-629b81a01a6b","resolution":{"observed_at":"2026-08-07T14:39:49.370355Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-07T14:24:42.605024Z","title":"Identify","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19213","last_updated":"2025-05-25T16:20:55Z","snapshot_observed_at":"2026-08-19T19:09:58.722542Z","submitted_at":"2025-05-25T16:20:55Z","title":"Improving Medical Reasoning with Curriculum-Aware Reinforcement Learning","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T14:24:42.605024Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2505.19213"},"observation_digest":"sha256:1731d68bef4c06e739ed0a24071746c42e1f1a83ef985589f4ca1d9a70da0476","observation_id":"247bc90c-215b-453a-ad87-185f4f606d53","resolution":{"observed_at":"2026-08-07T14:24:42.605024Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-07T14:21:42.988989Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19259","last_updated":"2025-05-28T02:16:52Z","snapshot_observed_at":"2026-08-21T19:59:17.516791Z","submitted_at":"2025-05-25T18:28:12Z","title":"Towards Large Reasoning Models for Agriculture","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T14:21:42.988989Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2505.19259"},"observation_digest":"sha256:a4930ebe19d680f90876af1d91f4b5b03649fc570d28f14906022609db759f16","observation_id":"7f4aa9d2-33be-4bf6-8def-c628972511ad","resolution":{"observed_at":"2026-08-07T14:21:42.988989Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-07T13:33:43.187164Z","title":"MedXpertQA: Benchmarking expert-level medical reasoning and understanding","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.21503","last_updated":"2025-05-27T17:59:50Z","snapshot_observed_at":"2026-08-14T15:15:23.898416Z","submitted_at":"2025-05-27T17:59:50Z","title":"Silence is Not Consensus: Disrupting Agreement Bias in Multi-Agent LLMs via Catfish Agent for Clinical Decision Making","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T13:33:43.187164Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2505.21503"},"observation_digest":"sha256:8a0311616db38f1597e6c686d9603e8d3b08efa6c756bd5a9954551bfee05357","observation_id":"a4d01c4d-8b62-4bf9-a0f4-2baa36503564","resolution":{"observed_at":"2026-08-07T13:33:43.187164Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-07T12:57:58.191080Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Rea- soning and Understanding","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.23075","last_updated":"2025-06-20T18:24:46Z","snapshot_observed_at":"2026-08-18T02:53:22.454989Z","submitted_at":"2025-05-29T04:29:22Z","title":"Second Opinion Matters: Towards Adaptive Clinical AI via the Consensus of Expert Model Ensemble","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T12:57:58.191080Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2505.23075"},"observation_digest":"sha256:35d2946bbeaa0437fb780ac0087688d58c2473a0998054e7c4886d2fb92eba45","observation_id":"361fc017-c4d4-4484-a06f-d00e895cb5f3","resolution":{"observed_at":"2026-08-07T12:57:58.191080Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-07T12:42:52.565185Z","title":"high relevance,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.24040","last_updated":"2025-05-29T22:23:48Z","snapshot_observed_at":"2026-08-17T22:44:18.037726Z","submitted_at":"2025-05-29T22:23:48Z","title":"MedPAIR: Measuring Physicians and AI Relevance Alignment in Medical Question Answering","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-07T12:42:52.565185Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2505.24040"},"observation_digest":"sha256:58b01e8e1f953b35b0aefff81aeafdbf677d95e28b3164075c290519fe45a66d","observation_id":"f548da07-6d9c-438d-bc51-f2528ac72cc9","resolution":{"observed_at":"2026-08-07T12:42:52.565185Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-07T12:05:36.112254Z","title":"URL https://www.biorxiv.org/content/ early/2023/10/17/2023.10.13.562216","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.00612","last_updated":"2025-07-03T16:50:12Z","snapshot_observed_at":"2026-08-15T10:54:16.886787Z","submitted_at":"2025-05-31T15:51:09Z","title":"Enhancing Clinical Multiple-Choice Questions Benchmarks with Knowledge Graph Guided Distractor Generation","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T12:05:36.112254Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2506.00612"},"observation_digest":"sha256:2bee93a9d510342ee10eb6621c6eb54255edef9150297ac7ae9d4d4de6e7c0b5","observation_id":"b373415e-d784-4a85-8e22-d02a0c3ab2c4","resolution":{"observed_at":"2026-08-07T12:05:36.112254Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-07T11:50:12.658854Z","title":"arXiv preprint arXiv:2501.18362, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.01257","last_updated":"2025-06-02T02:17:04Z","snapshot_observed_at":"2026-08-14T21:01:21.268956Z","submitted_at":"2025-06-02T02:17:04Z","title":"DeepSeek in Healthcare: A Survey of Capabilities, Risks, and Clinical Applications of Open-Source Large Language Models","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T11:50:12.658854Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2506.01257"},"observation_digest":"sha256:4081ccde740bfa53a7b70ea333f2db023dd9e178cfc3be1f8464f73ff4d7b4ba","observation_id":"5851fe81-99b1-4ede-8416-5cae63b5bf82","resolution":{"observed_at":"2026-08-07T11:50:12.658854Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-07T11:33:34.809777Z","title":"Medxpertqa: Benchmarking expert-level medical reasoning and understanding.arXiv preprint arXiv:2501.18362, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.02126","last_updated":"2025-06-02T18:01:00Z","snapshot_observed_at":"2026-08-16T15:42:44.759793Z","submitted_at":"2025-06-02T18:01:00Z","title":"Knowledge or Reasoning? A Close Look at How LLMs Think Across Domains","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T11:33:34.809777Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2506.02126"},"observation_digest":"sha256:316b29bdcabfbb1eba19d524577d2c941629e6e6b6bd42564017d4619425f4e7","observation_id":"b88cc0e5-d93c-4fef-8a9f-90ed663e0d5e","resolution":{"observed_at":"2026-08-07T11:33:34.809777Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-07T00:59:15.260193Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.12307","last_updated":"2025-06-20T01:43:46Z","snapshot_observed_at":"2026-08-15T14:50:44.575316Z","submitted_at":"2025-06-14T02:00:36Z","title":"Med-U1: Incentivizing Unified Medical Reasoning in LLMs via Large-scale Reinforcement Learning","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-07T00:59:15.260193Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2506.12307"},"observation_digest":"sha256:b457955752d4ad57a3391f036580cbc62f886f124af2768898a41f1c448a10ce","observation_id":"b21cae6e-9d43-4d4a-81f2-714733e39145","resolution":{"observed_at":"2026-08-07T00:59:15.260193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-15T20:08:56.122012Z","title":"MedXpertQA: Benchmarking expert-level medical reasoning and understanding,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.13102","last_updated":"2025-06-16T05:15:53Z","snapshot_observed_at":"2026-08-21T20:57:48.689137Z","submitted_at":"2025-06-16T05:15:53Z","title":"Rethinking Test-Time Scaling for Medical AI: Model and Task-Aware Strategies for LLMs and VLMs","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-15T20:08:56.122012Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2506.13102"},"observation_digest":"sha256:5891b109029c63711083e0282d0d7323ba2a4bff1f41cf963ba43e704a1c2d4f","observation_id":"79acf541-ad73-46d9-9e9d-714dce4bdc6e","resolution":{"observed_at":"2026-08-15T20:08:56.122012Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-06T23:10:43.721022Z","title":"Medxpertqa: Benchmarking expert-level medical reasoning and understanding.arXiv preprint arXiv:2501.18362, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.19217","last_updated":"2025-06-24T00:51:03Z","snapshot_observed_at":"2026-08-22T00:18:48.893924Z","submitted_at":"2025-06-24T00:51:03Z","title":"MedErr-CT: A Visual Question Answering Benchmark for Identifying and Correcting Errors in CT Reports","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T23:10:43.721022Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2506.19217"},"observation_digest":"sha256:13854d8b2e16851ccac8c15d2c9e8ef28ee6645e70fa0aaa652d933489903496","observation_id":"05f1c7c1-6992-4164-925c-2b7b3d5f5976","resolution":{"observed_at":"2026-08-06T23:10:43.721022Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2507.20917","last_updated":"2025-07-28T15:17:48Z","snapshot_observed_at":"2026-08-17T06:06:00.837915Z","submitted_at":"2025-07-28T15:17:48Z","title":"MediQAl: A French Medical Question Answering Dataset for Knowledge and Reasoning Evaluation","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-21T23:10:38.936021Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2507.20917"},"observation_digest":"sha256:147d2b9c4b97acb531d2e114d6f09208c4825dfc9cc472edce118b6d4c1fc12e","observation_id":"d22a01de-73f5-42e5-a047-00fe398d3a8f","resolution":{"observed_at":"2026-05-21T23:10:44.659252Z","resolver_source":"local_arxiv","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-06T11:45:40.971770Z","title":"Medxpertqa: Benchmarking expert-level medical reasoning and understanding","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-20T18:31:13.081515Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:40.971770Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:17ebcd1f7feed113e4e11ae4ad819818247198552775b67b27a74aa2a52fddfb","observation_id":"0c6493ec-dbf9-4033-8a9e-b6e045fa6b1d","resolution":{"observed_at":"2026-08-06T11:45:40.971770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-06T05:37:35.311823Z","title":"Medxpertqa: Benchmarking expert-level medical reasoning and understanding.arXiv preprint arXiv:2501.18362, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.01453","last_updated":"2025-08-02T18:02:44Z","snapshot_observed_at":"2026-08-08T09:26:45.590134Z","submitted_at":"2025-08-02T18:02:44Z","title":"Kernel-Based Sparse Additive Nonlinear Model Structure Detection through a Linearization Approach","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T05:37:35.311823Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2508.01453"},"observation_digest":"sha256:d70b497b862ac6dc16215dc0bd028a95b3f7310cd074a326d939e7c9d98079ab","observation_id":"8297b2ac-d6e7-40a3-8f9e-69e7730b7534","resolution":{"observed_at":"2026-08-06T05:37:35.311823Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T22:15:09.811969Z","title":"Medxpertqa: Benchmarking expert-level medical reasoning and understanding","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.07308","last_updated":"2025-08-10T11:45:34Z","snapshot_observed_at":"2026-08-22T03:21:32.140001Z","submitted_at":"2025-08-10T11:45:34Z","title":"HealthBranches: Synthesizing Clinically-Grounded Question Answering Datasets via Decision Pathways","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-05T22:15:09.811969Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2508.07308"},"observation_digest":"sha256:936111d3bec3a189aa656128ac4a4e5eb0c7314d5828362a353aa71c8abe5014","observation_id":"095c61c5-541d-402c-bfe5-81fd8a7f004f","resolution":{"observed_at":"2026-08-05T22:15:09.811969Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T21:38:48.110957Z","title":"Medxpertqa: Benchmarking expert-level medical reasoning and understanding","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.08224","last_updated":"2025-08-13T05:32:22Z","snapshot_observed_at":"2026-08-14T05:51:43.534966Z","submitted_at":"2025-08-11T17:43:45Z","title":"Capabilities of GPT-5 on Multimodal Medical Reasoning","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-05T21:38:48.110957Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2508.08224"},"observation_digest":"sha256:5db1bc7782037f058f18ad2ffa9d31202fa18bfea1b79063b84fdda4a5e6e60d","observation_id":"f8be9093-1622-40c3-9f40-0cdeebdd0e72","resolution":{"observed_at":"2026-08-05T21:38:48.110957Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-04T20:20:37.012685Z","title":"Medxpertqa: Benchmarking expert-level medical reasoning and understanding.arXiv preprint arXiv:2501.18362, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.08621","last_updated":"2025-09-10T14:17:53Z","snapshot_observed_at":"2026-08-18T02:51:18.202069Z","submitted_at":"2025-09-10T14:17:53Z","title":"AdsQA: Towards Advertisement Video Understanding","version":1},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-04T20:20:37.012685Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2509.08621"},"observation_digest":"sha256:138ffc7891d07673e38fb6d8c3b04f9361260864c2e369a3f1a67fa4deac726b","observation_id":"85fbb256-4269-4d16-b2cd-7529460147fe","resolution":{"observed_at":"2026-08-04T20:20:37.012685Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2509.24186","last_updated":"2026-04-06T07:24:03Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-09-29T02:06:13Z","title":"Measuring Competency, Not Performance: Item-Aware Evaluation Across Medical Benchmarks","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-05-18T13:16:19.744864Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2509.24186"},"observation_digest":"sha256:ef8228859299f146c15b23567d9369dfdf310542b804f3f447002bd2adb52e8b","observation_id":"3c9f4076-f74a-45f2-8fea-44f78b386640","resolution":{"observed_at":"2026-05-18T13:16:24.067374Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-04T13:51:53.456752Z","title":"Medxpertqa: Benchmarking expert-level medical reasoning and understanding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.24560","last_updated":"2026-08-02T06:52:09Z","snapshot_observed_at":"2026-08-19T05:36:29.932304Z","submitted_at":"2025-09-29T10:13:55Z","title":"AdaThink-Med: Optimizing Inference-Time Compute for Medical Reasoning via Uncertainty Quantification","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-04T13:51:53.456752Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2509.24560"},"observation_digest":"sha256:aefba6e1e3ba2fb9fe71d056ccf45a25ecb1b2810b0e7437ff749825f8a1f756","observation_id":"00f9f1d0-c640-4be0-9a55-37921d3b8350","resolution":{"observed_at":"2026-08-04T13:51:53.456752Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2602.07529","last_updated":"2026-04-15T23:53:20Z","snapshot_observed_at":"2026-08-13T14:53:00.086621Z","submitted_at":"2026-02-07T12:54:01Z","title":"MedVerse: Efficient and Reliable Medical Reasoning via DAG-Structured Parallel Execution","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-16T06:21:29.499231Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2602.07529"},"observation_digest":"sha256:3394e7ec74e27af17fb5b2d3d30034625a3dc953c0ee2954782eee672f9a4cf5","observation_id":"4c3a3772-61bb-422c-9f92-44370eb6e772","resolution":{"observed_at":"2026-05-16T14:26:38.603695Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2602.12705","last_updated":"2026-04-07T11:35:36Z","snapshot_observed_at":"2026-07-06T22:45:44.135338Z","submitted_at":"2026-02-13T08:19:38Z","title":"MedXIAOHE: A Comprehensive Recipe for Building Medical MLLMs","version":4},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-05-15T22:52:30.992054Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2602.12705"},"observation_digest":"sha256:d156b717a3efc7a119597d50f4cb79185c0d951d94d43bdcf1effba6c2b6acc6","observation_id":"d682ce68-1629-4254-b1eb-e48346c3f9d1","resolution":{"observed_at":"2026-05-16T14:26:38.603695Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-02T17:25:49.913986Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.25821","last_updated":"2026-08-13T17:57:32Z","snapshot_observed_at":"2026-08-19T01:24:25.440476Z","submitted_at":"2026-03-26T18:38:25Z","title":"Doctorina MedBench: A Dialogue-Based Benchmark and Evaluation Framework for Agent-Based Medical AI","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-02T17:25:49.913986Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2603.25821"},"observation_digest":"sha256:7a01f34a60696dba3b1b1882ea270ad5c1c25e37c9cef44b1e17414f5ecf0b0f","observation_id":"9610ecc2-1b58-4e3c-940b-02d187915dc1","resolution":{"observed_at":"2026-08-02T17:25:49.913986Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-02T17:19:40.131720Z","title":"Medxpertqa: Benchmarking expert-level medical reasoning and understanding.arXiv preprint arXiv:2501.18362, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.27176","last_updated":"2026-07-28T06:00:54Z","snapshot_observed_at":"2026-08-17T03:25:03.024800Z","submitted_at":"2026-03-28T07:26:40Z","title":"MEDIC-AD: Towards Medical Vision-Language Model's Clinical Intelligence","version":2},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-02T17:19:40.131720Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2603.27176"},"observation_digest":"sha256:1a6de5830b131d6491baaa293e44d2dbbcd049f6b95d84939a8a68068e2b055e","observation_id":"8d58c0f0-32c1-481f-911f-b7d8b8dab98a","resolution":{"observed_at":"2026-08-02T17:19:40.131720Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2604.05081","last_updated":"2026-05-01T19:02:06Z","snapshot_observed_at":"2026-08-12T16:02:17.136970Z","submitted_at":"2026-04-06T18:35:57Z","title":"MedGemma 1.5 Technical Report","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-10T19:36:41.167684Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2604.05081"},"observation_digest":"sha256:a7362c93b8597206cebc086c3c0b3bca3d515c58b61d67d73a61473a8444fa41","observation_id":"692ea18e-394a-496b-8778-b3574823c5d8","resolution":{"observed_at":"2026-05-16T14:26:38.603695Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2604.08644","last_updated":"2026-04-09T17:51:11Z","snapshot_observed_at":"2026-08-11T06:34:32.751399Z","submitted_at":"2026-04-09T17:51:11Z","title":"EXAONE 4.5 Technical Report","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-10T17:47:34.692414Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2604.08644"},"observation_digest":"sha256:75c8a25fc2e1cad922f8c431edb68442acc602cbe8769ccdb992f5592098b4e8","observation_id":"7ae76ceb-4b9c-4968-bbd5-44ef72b8a87f","resolution":{"observed_at":"2026-05-16T14:26:38.603695Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2604.09757","last_updated":"2026-07-19T04:58:08Z","snapshot_observed_at":"2026-08-14T07:00:07.053522Z","submitted_at":"2026-04-10T16:03:03Z","title":"MedLVR: Latent Visual Reasoning for Reliable Medical Visual Question Answering","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-10T18:03:14.408704Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2604.09757"},"observation_digest":"sha256:5aa1a38a1c5d9a421b2b28fed90933c4c5f0faf9e57aa8ed8165031c4325a008","observation_id":"b660a08e-c5e1-4234-a24f-06cbd6a39e56","resolution":{"observed_at":"2026-05-16T14:26:38.603695Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-02T16:33:56.516696Z","title":"Medxpertqa: Benchmarking expert-level medical reasoning and understanding,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2604.09757","last_updated":"2026-07-19T04:58:08Z","snapshot_observed_at":"2026-08-14T07:00:07.053522Z","submitted_at":"2026-04-10T16:03:03Z","title":"MedLVR: Latent Visual Reasoning for Reliable Medical Visual Question Answering","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-02T16:33:56.516696Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2604.09757"},"observation_digest":"sha256:1d1f506acfd812e6f1a47a1006e9ce4b95b8eda6d0a16478858ab8d27a25374b","observation_id":"45d2525e-cacb-4566-8379-fd6343f029f8","resolution":{"observed_at":"2026-08-02T16:33:56.516696Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2604.13756","last_updated":"2026-04-15T11:41:20Z","snapshot_observed_at":"2026-08-15T21:44:56.517909Z","submitted_at":"2026-04-15T11:41:20Z","title":"MedRCube: A Multidimensional Framework for Fine-Grained and In-Depth Evaluation of MLLMs in Medical Imaging","version":1},"reference_index":85,"source":"arxiv_source","source_observed_at":"2026-05-10T12:47:27.551670Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2604.13756"},"observation_digest":"sha256:0c7bc0ee5dacbfdd522bb8106bb9721005785415f76fed4059bda484c6616863","observation_id":"13da444f-16bc-453a-8cb0-333e43f456fc","resolution":{"observed_at":"2026-05-16T14:26:38.603695Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2604.17928","last_updated":"2026-04-20T08:09:01Z","snapshot_observed_at":"2026-08-14T00:24:03.820656Z","submitted_at":"2026-04-20T08:09:01Z","title":"HEALing Entropy Collapse: Enhancing Exploration in Few-Shot RLVR via Hybrid-Domain Entropy Dynamics Alignment","version":1},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-05-10T05:23:08.478393Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2604.17928"},"observation_digest":"sha256:050f27df10dd3718793d678370e76bb06afd87b640ad638af388060f33f09f65","observation_id":"68a546cd-ca2a-48e1-9666-add64b0b2e6c","resolution":{"observed_at":"2026-05-16T14:26:38.603695Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2604.19341","last_updated":"2026-04-21T11:24:09Z","snapshot_observed_at":"2026-08-13T08:23:53.899364Z","submitted_at":"2026-04-21T11:24:09Z","title":"Evaluation-driven Scaling for Scientific Discovery","version":1},"reference_index":178,"source":"pdf_text","source_observed_at":"2026-05-10T03:39:52.204043Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2604.19341"},"observation_digest":"sha256:6713d2a9f6a47abbc96934b975a1db5071134ad50bb077320cea7a1572e53770","observation_id":"dfd5bf8c-66ef-4eae-b0f0-f248f94f6c7b","resolution":{"observed_at":"2026-05-16T14:26:38.603695Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2604.24700","last_updated":"2026-04-27T17:04:17Z","snapshot_observed_at":"2026-08-11T03:23:03.664761Z","submitted_at":"2026-04-27T17:04:17Z","title":"Green Shielding: A User-Centric Approach Towards Trustworthy AI","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-08T03:43:54.896449Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2604.24700"},"observation_digest":"sha256:5f0a7510418c2a2fd3d5e3c4d654a3495bdf409f8b04417a034b9edfed7c06a5","observation_id":"3ea0346c-9da3-4694-b66a-712a2985cd70","resolution":{"observed_at":"2026-05-16T14:26:38.603695Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2604.26283","last_updated":"2026-07-02T12:22:56Z","snapshot_observed_at":"2026-08-11T12:22:37.659078Z","submitted_at":"2026-04-29T04:23:35Z","title":"MedSynapse-V: Bridging Visual Perception and Clinical Intuition via Latent Memory Evolution","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-07T14:02:54.395566Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2604.26283"},"observation_digest":"sha256:35e47f8cf2aaaf73743c840e6a6e8e19e53c56a326f8d6f4cfb42bfb630c412b","observation_id":"50884898-2721-41f2-82d6-5828cf462e2f","resolution":{"observed_at":"2026-05-16T14:26:38.603695Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2604.26283","last_updated":"2026-07-02T12:22:56Z","snapshot_observed_at":"2026-08-11T12:22:37.659078Z","submitted_at":"2026-04-29T04:23:35Z","title":"MedSynapse-V: Bridging Visual Perception and Clinical Intuition via Latent Memory Evolution","version":2},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-21T00:50:15.010488Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2604.26283"},"observation_digest":"sha256:e7f5fdde50eff2c48ecddd9470809f0cc783421c20d484b2d2f512691a2f6bce","observation_id":"9298d93c-375f-449f-abef-a1fdf5de2c7d","resolution":{"observed_at":"2026-05-21T00:53:53.394460Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2604.26283","last_updated":"2026-07-02T12:22:56Z","snapshot_observed_at":"2026-08-11T12:22:37.659078Z","submitted_at":"2026-04-29T04:23:35Z","title":"MedSynapse-V: Bridging Visual Perception and Clinical Intuition via Latent Memory Evolution","version":3},"reference_index":161,"source":"pdf_text","source_observed_at":"2026-07-01T09:10:28.244690Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2604.26283"},"observation_digest":"sha256:4dce3451b7b3bfb77037770985a1b52e51a2a7f16a71b08319809c3b14512149","observation_id":"2ba01998-58db-4f36-a07f-b138d6a680a6","resolution":{"observed_at":"2026-07-01T09:15:43.776509Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2604.26283","last_updated":"2026-07-02T12:22:56Z","snapshot_observed_at":"2026-08-11T12:22:37.659078Z","submitted_at":"2026-04-29T04:23:35Z","title":"MedSynapse-V: Bridging Visual Perception and Clinical Intuition via Latent Memory Evolution","version":4},"reference_index":161,"source":"pdf_text","source_observed_at":"2026-07-04T01:35:11.966506Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2604.26283"},"observation_digest":"sha256:c30588c79ab16c3e3763f80c373e1a81785a8df3e2a1ea4511a630a8248283fb","observation_id":"48872420-c0a9-49c1-a1b6-42bb76b17d3b","resolution":{"observed_at":"2026-07-04T01:39:23.926310Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2605.06191","last_updated":"2026-05-07T13:05:07Z","snapshot_observed_at":"2026-07-06T23:18:41.400741Z","submitted_at":"2026-05-07T13:05:07Z","title":"Systematic Evaluation of Large Language Models for Post-Discharge Clinical Action Extraction","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-08T10:15:03.503011Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2605.06191"},"observation_digest":"sha256:d3f7b5bfbbfd49eefc9e85906e3b1af843e8f727552e404dfcc5e2e62b1318af","observation_id":"54dcbd29-0610-4074-a560-598177c4e162","resolution":{"observed_at":"2026-05-16T14:26:38.603695Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2605.08704","last_updated":"2026-08-10T03:56:56Z","snapshot_observed_at":"2026-08-13T23:39:11.393391Z","submitted_at":"2026-05-09T05:38:21Z","title":"AgentPSO: Evolving Agent Reasoning Skill via Multi-agent Particle Swarm Optimization","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-12T00:55:52.965890Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2605.08704"},"observation_digest":"sha256:d33c7737cc73168d7b798dce94b8c695712df1ab415b6eee3ac95763c48ea7d5","observation_id":"c3f0f6c6-a27d-476c-a3b3-c2901221c1d6","resolution":{"observed_at":"2026-05-16T14:26:38.603695Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2605.08704","last_updated":"2026-08-10T03:56:56Z","snapshot_observed_at":"2026-08-13T23:39:11.393391Z","submitted_at":"2026-05-09T05:38:21Z","title":"AgentPSO: Evolving Agent Reasoning Skill via Multi-agent Particle Swarm Optimization","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-06-30T23:29:12.545347Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2605.08704"},"observation_digest":"sha256:cb73b45940e8564c76f8a227964f8c2c54c098991b171af8e5904ebc6c182ed5","observation_id":"32adff1c-b1b1-4a8b-80d5-bc214f8c2403","resolution":{"observed_at":"2026-06-30T23:35:07.613373Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2605.09584","last_updated":"2026-05-10T14:51:31Z","snapshot_observed_at":"2026-07-06T23:21:39.966502Z","submitted_at":"2026-05-10T14:51:31Z","title":"CLR-voyance: Reinforcing Open-Ended Reasoning for Inpatient Clinical Decision Support with Outcome-Aware Rubrics","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-05-12T04:32:16.930291Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2605.09584"},"observation_digest":"sha256:21d90e2c9a7621f9babf5104e1507ec6cc040184bd2e5680012280ab8d7fff8d","observation_id":"4212d254-c39a-40c9-9d29-38092cc7b065","resolution":{"observed_at":"2026-05-16T14:26:38.603695Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2605.09661","last_updated":"2026-05-10T17:20:39Z","snapshot_observed_at":"2026-08-02T16:02:49.491231Z","submitted_at":"2026-05-10T17:20:39Z","title":"MedMeta: A Benchmark for LLMs in Synthesizing Meta-Analysis Conclusion from Medical Studies","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-12T03:37:22.985929Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2605.09661"},"observation_digest":"sha256:3e7c517064f69d221fa45fea6290e2a0103cfc40f1d3acc0b3d023a0b6911ba6","observation_id":"672cb521-246b-48bd-9dca-de6c0991ecbe","resolution":{"observed_at":"2026-05-16T14:26:38.603695Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2605.10850","last_updated":"2026-05-11T17:00:00Z","snapshot_observed_at":"2026-08-13T23:36:40.689739Z","submitted_at":"2026-05-11T17:00:00Z","title":"Verification Mirage: Mapping the Reliability Boundary of Self-Verification in Medical VQA","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-12T04:48:09.071812Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2605.10850"},"observation_digest":"sha256:9e256bc9f60cafb24c0fae8020c7146e928cc2fce3dca924d6c6144f2aa32bb8","observation_id":"c045fad3-5bf3-4edd-80fc-4c8670828acc","resolution":{"observed_at":"2026-05-16T14:26:38.603695Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2605.13045","last_updated":"2026-05-13T06:04:40Z","snapshot_observed_at":"2026-08-14T22:54:54.013861Z","submitted_at":"2026-05-13T06:04:40Z","title":"Large Language Models Lack Temporal Awareness of Medical Knowledge","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-14T20:18:08.160768Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2605.13045"},"observation_digest":"sha256:89ffcb83ad89a55b6378f08fe3b622718554103635c0b50c065b0205762ec5c8","observation_id":"dc6b6f94-f700-4f62-a434-6da05e27de3c","resolution":{"observed_at":"2026-05-16T14:26:38.603695Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2605.13542","last_updated":"2026-05-13T13:52:42Z","snapshot_observed_at":"2026-08-15T10:55:02.338637Z","submitted_at":"2026-05-13T13:52:42Z","title":"RealICU: Do LLM Agents Understand Long-Context ICU Data? A Benchmark Beyond Behavior Imitation","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-14T18:47:46.239810Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2605.13542"},"observation_digest":"sha256:686bf87b17ff1cdce6da45406f445a93cdbb212417f900a32ab77c4aaa8dc5bf","observation_id":"d65abc5c-b48d-43f6-a642-ed0e3d3b4e88","resolution":{"observed_at":"2026-05-16T14:26:38.603695Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2605.16215","last_updated":"2026-05-29T15:56:10Z","snapshot_observed_at":"2026-08-15T03:50:46.772880Z","submitted_at":"2026-05-15T17:29:08Z","title":"Fully Open Meditron: An Auditable Pipeline for Clinical LLMs","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-20T18:53:16.389210Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2605.16215"},"observation_digest":"sha256:9202ca2fbde671680d502a5ccb60250ff330247971296f90e6fee77ee6b1ed50","observation_id":"55812b3a-6954-483a-8deb-28937209982f","resolution":{"observed_at":"2026-05-20T18:53:38.771817Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2605.16215","last_updated":"2026-05-29T15:56:10Z","snapshot_observed_at":"2026-08-15T03:50:46.772880Z","submitted_at":"2026-05-15T17:29:08Z","title":"Fully Open Meditron: An Auditable Pipeline for Clinical LLMs","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-06-30T19:14:06.301012Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2605.16215"},"observation_digest":"sha256:2df523787787def8e3c34513bfd628622b54902dd943efc198c89ae9c9f0f816","observation_id":"59e7a0e2-c6b9-41cb-a6ec-1749cb4a4634","resolution":{"observed_at":"2026-06-30T19:15:00.117175Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2605.16679","last_updated":"2026-05-19T05:51:20Z","snapshot_observed_at":"2026-08-16T12:26:18.226673Z","submitted_at":"2026-05-15T22:34:31Z","title":"CHI-Bench: Can AI Agents Automate End-to-End, Long-Horizon, Policy-Rich Healthcare Workflows?","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-05-20T17:45:02.896703Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2605.16679"},"observation_digest":"sha256:d9f58aa2d8f3d586671007500b9f34689e59e4c7f65990676ba7a15b88cc5c21","observation_id":"95391074-b241-441b-83ca-749b1519187f","resolution":{"observed_at":"2026-05-20T17:48:48.822438Z","resolver_source":"local_arxiv","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2605.20176","last_updated":"2026-05-19T17:58:37Z","snapshot_observed_at":"2026-08-15T16:55:48.940165Z","submitted_at":"2026-05-19T17:58:37Z","title":"ClinSeekAgent: Automating Multimodal Evidence Seeking for Agentic Clinical Reasoning","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-20T05:19:35.222068Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2605.20176"},"observation_digest":"sha256:0291b4288c6ced86a282d6112b2c5ecf51fa023944b4e05b98e1d5f9fda2bb2a","observation_id":"ca86765d-829e-4960-845c-477932d03578","resolution":{"observed_at":"2026-05-20T05:23:03.760948Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2605.20525","last_updated":"2026-05-19T21:54:12Z","snapshot_observed_at":"2026-07-06T23:31:08.671011Z","submitted_at":"2026-05-19T21:54:12Z","title":"NeuroQA: A Large-Scale Image-Grounded Benchmark for 3D Brain MRI Understanding","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-21T06:54:55.254082Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2605.20525"},"observation_digest":"sha256:415c2a5be89fd6f72766015375e59aa1cffd1a45154db71a369fbb85add08f58","observation_id":"33e9182b-8544-4f99-8121-9d884276b0b6","resolution":{"observed_at":"2026-05-21T06:59:45.784825Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2605.23629","last_updated":"2026-05-22T13:41:10Z","snapshot_observed_at":"2026-08-16T11:12:24.170172Z","submitted_at":"2026-05-22T13:41:10Z","title":"DDX-TRACE: A Benchmark for Medical Diagnostic Trajectories in VLMs","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-25T04:53:33.343509Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2605.23629"},"observation_digest":"sha256:9a5b67e2b3311efdef73c520778a689b35427cff0940111d7087597a952a31f8","observation_id":"c85c7064-5477-457d-8fcb-e18e82f47da0","resolution":{"observed_at":"2026-05-25T04:55:23.447614Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2605.27860","last_updated":"2026-08-03T08:40:06Z","snapshot_observed_at":"2026-08-14T11:10:07.093303Z","submitted_at":"2026-05-27T02:20:21Z","title":"C-MIG: Multi-view Information Gain-based Retrieval-Augmented Generation for Clinical Diagnosis Reasoning","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-29T12:43:58.323948Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2605.27860"},"observation_digest":"sha256:127f7fc767dffe02dd703a2b0f4d61258f8ed47505a3cc60238754cf542bd2d1","observation_id":"7816dc5c-541f-4b18-b5eb-35b694a079b3","resolution":{"observed_at":"2026-06-29T12:53:27.009199Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-04T05:01:11.077901Z","title":"arXiv preprint arXiv:2501.18362","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2605.27860","last_updated":"2026-08-03T08:40:06Z","snapshot_observed_at":"2026-08-14T11:10:07.093303Z","submitted_at":"2026-05-27T02:20:21Z","title":"C-MIG: Multi-view Information Gain-based Retrieval-Augmented Generation for Clinical Diagnosis Reasoning","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-04T05:01:11.077901Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2605.27860"},"observation_digest":"sha256:6675d5e82868021861c1e3921a3c79c1f44fe0cf68c46cb0ab9a9d03d210ef7a","observation_id":"719eb01f-bf4f-4d10-ad0e-04ef4c9b6da5","resolution":{"observed_at":"2026-08-04T05:01:11.077901Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2605.28631","last_updated":"2026-05-27T15:38:09Z","snapshot_observed_at":"2026-08-16T21:35:12.845485Z","submitted_at":"2026-05-27T15:38:09Z","title":"Single-Rollout Hidden-State Dynamics for Training-Free RLVR Data Selection","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-29T14:08:40.968105Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2605.28631"},"observation_digest":"sha256:ac069be022950bde1eeb90ac2af82c591e1b8c5651ff0aac55aa533987d4ef96","observation_id":"a7275da5-de27-4e80-8de7-df8efc98f7d5","resolution":{"observed_at":"2026-06-29T14:13:30.113213Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2605.30637","last_updated":"2026-05-28T22:38:26Z","snapshot_observed_at":"2026-08-15T02:58:19.749294Z","submitted_at":"2026-05-28T22:38:26Z","title":"EHRBench: An Automated and Reliable EHR-based Benchmark for Clinical Decision Making with LLMs","version":1},"reference_index":132,"source":"pdf_text","source_observed_at":"2026-06-29T06:41:06.828814Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2605.30637"},"observation_digest":"sha256:225ea632ab55f72b11e3f2c2eb54a6f404c556d60bb75a2c953b11563db4753f","observation_id":"d2c790ea-3813-48b5-9379-580a9ed8a723","resolution":{"observed_at":"2026-06-29T06:43:10.362776Z","resolver_source":"local_arxiv","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2606.02082","last_updated":"2026-06-01T11:12:28Z","snapshot_observed_at":"2026-08-14T14:00:50.706907Z","submitted_at":"2026-06-01T11:12:28Z","title":"Overview of the ClinicalSkillQA 2026 Shared Task on Continuous Perception and Procedural Reasoning in Clinical Skill Assessment","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-06-28T12:56:39.071730Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2606.02082"},"observation_digest":"sha256:44efd5c222c597e8e4554d01470d8c668c42e9bf240924e06b984acf324e8df0","observation_id":"49accd18-ae0e-4825-83b6-1fce3e5503ad","resolution":{"observed_at":"2026-07-02T00:56:25.600920Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2606.05112","last_updated":"2026-06-03T17:17:16Z","snapshot_observed_at":"2026-07-06T23:45:06.925235Z","submitted_at":"2026-06-03T17:17:16Z","title":"Evaluating Large Language Models in Dynamic Clinical Decision-Making with Standardized Patient Cases","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-28T06:33:50.348031Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2606.05112"},"observation_digest":"sha256:a3bfcec24216c25a476c7e4e0787d3e36bad2fa917c08233ccc14befecb67d45","observation_id":"44c24813-c531-46d1-8cdf-d36d4352ab14","resolution":{"observed_at":"2026-07-02T07:56:47.197661Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2606.06840","last_updated":"2026-06-05T02:32:24Z","snapshot_observed_at":"2026-08-16T20:02:22.679518Z","submitted_at":"2026-06-05T02:32:24Z","title":"Characterize Then Distill: Mechanistic Reasoning in Large Output Spaces","version":1},"reference_index":114,"source":"arxiv_source","source_observed_at":"2026-06-27T22:22:52.690010Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2606.06840"},"observation_digest":"sha256:2d51783f4edb68937c19afa11acfb0ddd945ec77a91306ef7c7dc37c32945380","observation_id":"d8aa4c76-c0f8-4783-b41f-891c3a40aebe","resolution":{"observed_at":"2026-06-27T22:31:21.528834Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2606.07549","last_updated":"2026-05-18T12:30:03Z","snapshot_observed_at":"2026-08-14T13:27:01.386790Z","submitted_at":"2026-05-18T12:30:03Z","title":"PathoSage: Towards Multi-Source Evidence Adjudication in Pathology via Experience-Aware Agentic Workflow","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-06-30T18:39:14.083052Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2606.07549"},"observation_digest":"sha256:8d0c8085835d0a299160c9b937716a0e8b8e6dba2e38032cdc13ef3b78d80b93","observation_id":"6c41d075-8dfc-44e4-9406-73c55ae184a3","resolution":{"observed_at":"2026-06-30T19:15:01.454125Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2606.07853","last_updated":"2026-06-05T21:29:39Z","snapshot_observed_at":"2026-08-16T08:17:11.618542Z","submitted_at":"2026-06-05T21:29:39Z","title":"Beyond English benchmarks: clinical llm evaluation in Brazilian Portuguese","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-27T21:40:16.052239Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2606.07853"},"observation_digest":"sha256:75d3b63b4f1c7def34aa0d77540e797b91175de4287ac2377c522302b99c6fb0","observation_id":"856940c0-4d29-44cf-8f71-bc9abe4e8525","resolution":{"observed_at":"2026-07-02T19:07:17.905485Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2606.08231","last_updated":"2026-06-06T15:39:29Z","snapshot_observed_at":"2026-08-07T16:58:05.550639Z","submitted_at":"2026-06-06T15:39:29Z","title":"Test-Time Scaling in Multimodal Foundation Models: A Comprehensive Survey of Generation and Reasoning","version":1},"reference_index":121,"source":"arxiv_source","source_observed_at":"2026-06-27T19:36:57.231932Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2606.08231"},"observation_digest":"sha256:8a3a908b5a4f4662dce9f34480b3bb1ef737131497ac51da3ec33f98354312c6","observation_id":"9e6b1dec-f544-42f4-a64f-aaf7f14127de","resolution":{"observed_at":"2026-07-02T21:37:25.362554Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2606.09365","last_updated":"2026-06-10T14:46:00Z","snapshot_observed_at":"2026-08-16T14:19:11.302905Z","submitted_at":"2026-06-08T11:37:01Z","title":"Experience Makes Skillful: Enabling Generalizable Medical Agent Reasoning via Self-Evolving Skill Memory","version":2},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-06-27T16:45:30.431403Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2606.09365"},"observation_digest":"sha256:49bc93028fc7d67f70fd246b46dd18611d39760a90c2aafae5462b3044e4c6b7","observation_id":"306a91b9-acf7-444a-bc11-08bb9a15bd19","resolution":{"observed_at":"2026-07-03T01:07:30.881909Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2606.11740","last_updated":"2026-06-10T07:16:27Z","snapshot_observed_at":"2026-08-02T02:17:16.809620Z","submitted_at":"2026-06-10T07:16:27Z","title":"UniReason-Med: A Shared Grounded Reasoning Interface for 2D-to-3D Transfer in Medical VQA","version":1},"reference_index":189,"source":"arxiv_source","source_observed_at":"2026-06-27T10:21:12.782864Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2606.11740"},"observation_digest":"sha256:e399d8b77e083b27d3eb6f9c28b9089b9e9ed336c11de79c258417c8ba81167c","observation_id":"a88060c2-2971-4941-94ae-5f5925a5127a","resolution":{"observed_at":"2026-07-03T09:47:59.439147Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2606.12291","last_updated":"2026-06-10T16:27:26Z","snapshot_observed_at":"2026-08-17T11:18:59.732430Z","submitted_at":"2026-06-10T16:27:26Z","title":"Measuring Epistemic Resilience of LLMs Under Misleading Medical Context","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-06-27T09:30:47.726480Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2606.12291"},"observation_digest":"sha256:44ed7b14ba1141089eb244b9e1f00fcfd2ba07f0bf4fdf9ee3d3ca96f6466012","observation_id":"85ae5e85-f44d-490e-8969-4645295422f7","resolution":{"observed_at":"2026-07-03T11:28:04.910150Z","resolver_source":"local_arxiv","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2606.19266","last_updated":"2026-06-17T16:42:22Z","snapshot_observed_at":"2026-08-06T06:32:08.880714Z","submitted_at":"2026-06-17T16:42:22Z","title":"Trade-offs in Medical LLM Adaptation: An Empirical Study in French QA","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-26T21:00:06.185627Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2606.19266"},"observation_digest":"sha256:411e6f7d196b8ffe1cbbd65ac3a5dc8fe043d80a9756ba2a38250df033455032","observation_id":"a7d1cc88-8241-470f-ae98-baecc0da6daf","resolution":{"observed_at":"2026-07-04T00:49:17.773387Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2606.21020","last_updated":"2026-06-19T01:10:24Z","snapshot_observed_at":"2026-08-12T23:33:42.994470Z","submitted_at":"2026-06-19T01:10:24Z","title":"CheXpercept: A Benchmark for Evaluating Expert-Level Lesion Perception in Chest X-rays","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-06-26T15:05:13.422803Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2606.21020"},"observation_digest":"sha256:01dbac26bae3057f0a29cc06a0be308c56206a3d10faac1215ab5337e1ab1d7f","observation_id":"23739f5c-0713-4072-a46d-d56d9764052a","resolution":{"observed_at":"2026-07-04T05:59:37.423945Z","resolver_source":"local_arxiv","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2606.21937","last_updated":"2026-06-20T08:13:31Z","snapshot_observed_at":"2026-08-14T10:54:41.360384Z","submitted_at":"2026-06-20T08:13:31Z","title":"Latent Confidence Alignment for LLM Self-Assessment","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-06-26T11:21:33.744610Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2606.21937"},"observation_digest":"sha256:a2287b231cd17f5e297975c88509e1caae916436b81d2e413ea03029bada9da1","observation_id":"6b6aaf1a-4d1f-4df2-a0bf-edf760cd9648","resolution":{"observed_at":"2026-07-04T08:39:41.809911Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2606.22437","last_updated":"2026-06-25T12:10:03Z","snapshot_observed_at":"2026-08-14T07:50:04.090458Z","submitted_at":"2026-06-21T10:57:43Z","title":"MMGist: A Comprehensive Multimodal Benchmark for 2027","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-26T11:05:14.573386Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2606.22437"},"observation_digest":"sha256:d4a5706523168d971419c82cbeef42a74e2549284e34dccf4da4313a54631be9","observation_id":"22abc74e-6ac9-46d8-bf11-d81713b57c31","resolution":{"observed_at":"2026-07-04T08:49:41.621498Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2606.26079","last_updated":"2026-06-24T17:53:26Z","snapshot_observed_at":"2026-08-20T06:49:45.555280Z","submitted_at":"2026-06-24T17:53:26Z","title":"Same Evidence, Different Answer: Auditing Order Sensitivity in Multimodal Large Language Models","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-06-25T19:58:23.594907Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2606.26079"},"observation_digest":"sha256:014f7ab4273d98ddf56bdd6929b73e413ad71065fa87934b837bdf204fc26b8a","observation_id":"886ab2bc-75ad-485a-80fe-6de71f878af1","resolution":{"observed_at":"2026-07-04T20:40:07.573527Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2606.26797","last_updated":"2026-06-25T09:32:58Z","snapshot_observed_at":"2026-08-12T19:45:04.345071Z","submitted_at":"2026-06-25T09:32:58Z","title":"Reasoning Quality Emerges Early: Data Curation for Reasoning Models","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-11T11:50:26.030339Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2606.26797"},"observation_digest":"sha256:1d5ef843150e5d0e2dafc9cf40aa0ca2b65a2263a29d765476d1eb6cd7e7046f","observation_id":"c645a31a-7ca4-43fc-b4c5-8ad9b304d649","resolution":{"observed_at":"2026-06-26T05:08:59.952542Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2606.31599","last_updated":"2026-06-30T12:47:30Z","snapshot_observed_at":"2026-07-07T00:05:17.259491Z","submitted_at":"2026-06-30T12:47:30Z","title":"Token-Sparse Medical Multimodal Reasoning via Dual-Stream Reinforcement Learning","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-07-01T05:36:43.609602Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2606.31599"},"observation_digest":"sha256:a49c5a59c86db3f1d45cb4132fcf88c639b7f710d38f6a36b4a53474130030f1","observation_id":"bc6b8fa0-2b7f-475c-ae5c-6a95e4b23536","resolution":{"observed_at":"2026-07-01T10:15:45.207200Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":"2501.18362","doi":"10.18653/v1/2025.acl-long.452","metadata_source":"pith","pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","venue":"cs.AI","work_id":"17099117-68ce-46a6-a057-ac254bc140ac","year":2025},"citing_paper":{"arxiv_id":"2607.01440","last_updated":"2026-07-01T20:02:55Z","snapshot_observed_at":"2026-07-07T00:06:59.129459Z","submitted_at":"2026-07-01T20:02:55Z","title":"FaithMed: Training LLMs For Faithful Evidence-Based Medical Reasoning","version":1},"reference_index":87,"source":"arxiv_source","source_observed_at":"2026-07-03T21:08:02.793814Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2607.01440"},"observation_digest":"sha256:29ca8b08b23da303c4a298644e65ae72ac5a63a3acbc6e4595773c0a49516429","observation_id":"910b5549-5573-4a10-b9e5-1d98a5bc533a","resolution":{"observed_at":"2026-07-03T21:08:57.474691Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-07-12T07:10:06.281832Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.02770","last_updated":"2026-07-02T21:08:53Z","snapshot_observed_at":"2026-08-18T16:22:37.329414Z","submitted_at":"2026-07-02T21:08:53Z","title":"Gemma 4 Technical Report","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-07-12T07:10:06.281832Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2607.02770"},"observation_digest":"sha256:2c17b1f772bbac9d570a3bdd8e58ef3c1641d8b129b318734cda004cc21ab3aa","observation_id":"299e3775-ebab-4eb0-b926-14ce590cb344","resolution":{"observed_at":"2026-07-12T07:10:06.281832Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-07-12T00:57:35.205227Z","title":"arXiv preprint arXiv:2501.18362 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.03647","last_updated":"2026-07-04T00:06:35Z","snapshot_observed_at":"2026-08-06T08:51:21.151719Z","submitted_at":"2026-07-04T00:06:35Z","title":"Do Medical Vision Language Models Actually See? A Counterfactual Grounding Framework and Hard-Negative Contrastive Training for Visually-Reliant Medical VLMs","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-07-12T00:57:35.205227Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2607.03647"},"observation_digest":"sha256:6049a80f700ce011940b5af6c30bea80787d29b37062d56c7ac07b4c45f9176c","observation_id":"f77e7e17-e43d-46e9-b61f-9c5ddbc01eb0","resolution":{"observed_at":"2026-07-12T00:57:35.205227Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-02T06:32:29.811544Z","title":"Medxpertqa: Benchmarking expert-level medical reasoning and understanding.arXiv preprint arXiv:2501.18362, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.12527","last_updated":"2026-07-21T09:07:43Z","snapshot_observed_at":"2026-08-21T17:19:44.324947Z","submitted_at":"2026-07-14T09:03:26Z","title":"Evidence-Grounded AI for Musculoskeletal Care","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-02T06:32:29.811544Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2607.12527"},"observation_digest":"sha256:a9f0236e936739c54aed945be0bc0f143cada2cd851e4ee4e2df2df0e09b5788","observation_id":"92c4a64b-125d-4981-87df-64f0aa6714ac","resolution":{"observed_at":"2026-08-02T06:32:29.811544Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-02T02:16:39.947485Z","title":"Medxpertqa: Benchmarking expert-level medical reasoning and understanding.arXiv preprint arXiv:2501.18362,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15314","last_updated":"2026-08-04T04:26:54Z","snapshot_observed_at":"2026-08-17T19:58:45.517557Z","submitted_at":"2026-07-15T22:05:23Z","title":"Cura 1T: Specialized Model for Agentic Healthcare","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-02T02:16:39.947485Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2607.15314"},"observation_digest":"sha256:5f15016a9c4481e03681c535c0d19f2d6ce176af043a013f18ad03c16aad9c23","observation_id":"8e42376f-5c2a-41f0-9723-1e65cace1f15","resolution":{"observed_at":"2026-08-02T02:16:39.947485Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-01T12:04:29.120119Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.19678","last_updated":"2026-07-22T02:27:51Z","snapshot_observed_at":"2026-08-20T08:51:45.822616Z","submitted_at":"2026-07-22T02:27:51Z","title":"Reference-Free Evaluation of Reasoning in Open-Ended Question Answering","version":1},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-01T12:04:29.120119Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2607.19678"},"observation_digest":"sha256:eace4cb59f0eaf888550c20c08ab3e6b848f843b25824897775c09c88c195cde","observation_id":"1a18290f-fb69-4e97-90c3-c689dd7a6708","resolution":{"observed_at":"2026-08-01T12:04:29.120119Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-02T13:52:07.924487Z","title":"MedXpertQA: Benchmarking expert-level medical reasoning and understanding","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.20462","last_updated":"2026-05-16T08:16:08Z","snapshot_observed_at":"2026-08-18T21:40:44.057807Z","submitted_at":"2026-05-16T08:16:08Z","title":"Marking the Wrong Symptoms: Evaluating LLM Watermarks in Medical Texts","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-02T13:52:07.924487Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2607.20462"},"observation_digest":"sha256:17dfe4c6193c3a1cf855c7102e485409ba8b073a1dacb52343a72ccf07971da9","observation_id":"91269e56-7bc3-4f7c-9000-d352c58b5e09","resolution":{"observed_at":"2026-08-02T13:52:07.924487Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-01T08:37:05.181089Z","title":"A Additional Experimental Results and Analyses In this section, we provide additional experimental results and analyses that complement the findings in the main pa- per","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21065","last_updated":"2026-07-23T08:59:11Z","snapshot_observed_at":"2026-08-13T15:22:42.804748Z","submitted_at":"2026-07-23T08:59:11Z","title":"Do Pathology Vision-Language Models Truly See Pathology?","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-01T08:37:05.181089Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2607.21065"},"observation_digest":"sha256:495adc666a65c3990b0797d69a2bd8caa795f63e5a9ad4a9833c0082b56f3403","observation_id":"0cc6bf25-81ce-4bee-b96c-b85c31023eb4","resolution":{"observed_at":"2026-08-01T08:37:05.181089Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-01T06:12:07.519917Z","title":"arXiv preprint arXiv:2501.18362 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.24838","last_updated":"2026-07-24T06:02:41Z","snapshot_observed_at":"2026-08-09T00:30:29.801498Z","submitted_at":"2026-07-24T06:02:41Z","title":"MedJudgeRAG: Option-Wise Evidence Judgment with Dynamic Knowledge Graphs for Medical MCQA","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-01T06:12:07.519917Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2607.24838"},"observation_digest":"sha256:3430e62bf98ec7e37afc162f649eb871a75ca0599704c39e023333681f926ada","observation_id":"e284e0ee-8709-43c8-a2da-e330f6077dea","resolution":{"observed_at":"2026-08-01T06:12:07.519917Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-01T02:21:21.671396Z","title":"Proceedings of the 42nd International Conference on Machine Learning (ICML) , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.25485","last_updated":"2026-07-28T09:24:04Z","snapshot_observed_at":"2026-08-14T06:59:07.280075Z","submitted_at":"2026-07-28T09:24:04Z","title":"PatientAgentBench: A Benchmark Framework for Evaluating Patient-Facing Health AI Agents","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-01T02:21:21.671396Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2607.25485"},"observation_digest":"sha256:f33e6f3de5245755143da270e0105409810f017a7374e2ca7f6c544cd996930b","observation_id":"81cfd0c9-d0ce-4fb6-a6a4-312c336c1c78","resolution":{"observed_at":"2026-08-01T02:21:21.671396Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-01T05:36:47.916051Z","title":"MedXpertQA: Benchmarking expert-level medical reasoning and understanding.arXiv preprint arXiv:2501.18362, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.27566","last_updated":"2026-07-30T01:19:28Z","snapshot_observed_at":"2026-08-15T14:29:56.265537Z","submitted_at":"2026-07-30T01:19:28Z","title":"Objective-Aligned Direct Answer SFT for Robust Multi-Frame Medical VQA","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-01T05:36:47.916051Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2607.27566"},"observation_digest":"sha256:90ce9241f14095611384af02c2e4dcf1fec71ced611f50aa276f22df4d8b7c81","observation_id":"b6a1f3c6-bcce-4ae2-a7fa-3721e66fa241","resolution":{"observed_at":"2026-08-01T05:36:47.916051Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-15T14:34:14.608714Z","title":"arXiv preprint arXiv:2501.18362 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.06867","last_updated":"2026-08-07T06:46:58Z","snapshot_observed_at":"2026-08-20T04:51:28.575652Z","submitted_at":"2026-08-07T06:46:58Z","title":"LLMRouter: Unified Infrastructure for Developing, Evaluating, and Deploying LLM Routers","version":1},"reference_index":206,"source":"arxiv_source","source_observed_at":"2026-08-15T14:34:14.608714Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2608.06867"},"observation_digest":"sha256:6d4e85985c8a8b3af5bd190926962f393b030df734339230fc00c2138df8cee5","observation_id":"d2990618-5134-47dd-948f-cd0808a2a1d8","resolution":{"observed_at":"2026-08-15T14:34:14.608714Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-12T16:38:14.279760Z","title":"name\": \"SEARCH","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10827","last_updated":"2026-08-11T11:57:27Z","snapshot_observed_at":"2026-08-19T20:38:23.113883Z","submitted_at":"2026-08-11T11:57:27Z","title":"MIRA: Medical Image Reflection for Agentic Diagnosis","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-12T16:38:14.279760Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2608.10827"},"observation_digest":"sha256:f581996b8ac9f5a769fbafd20cf4e8ab2294d53c67a973aa34dc2ce0eb0ce188","observation_id":"e68527a9-4ca7-46a3-b8f0-f5778ba0f14c","resolution":{"observed_at":"2026-08-12T16:38:14.279760Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2501.18362/citation-record","integrity":"/paper/2501.18362/integrity","json":"/paper/2501.18362/citation-record.json","paper":"/paper/2501.18362"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"org/CorpusID:268232499","venue":null,"work_id":"af9e8a01-2fb0-4264-b572-68c9d9f2b49d","year":null},"citing_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-16T14:26:38.517787Z"},"links":{"citing_paper":"/paper/2501.18362"},"observation_digest":"sha256:23f6c26791664baf3d6f32cc7dadf70578d1ba5dcd71308a0c14bf5497c3dfc8","observation_id":"1681f3dd-2327-4d38-a78c-18accf2e4868","resolution":{"observed_at":"2026-05-16T14:26:38.600620Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.18925","last_updated":"2024-12-25T15:12:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-25T15:12:34Z","title":"HuatuoGPT-o1, Towards Medical Complex Reasoning with LLMs","version":1},"cited_work":{"arxiv_id":"2412.18925","doi":"10.48550/arxiv.2412.18925","metadata_source":"pith","pith_arxiv_id":"2412.18925","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"HuatuoGPT-o1, Towards Medical Complex Reasoning with LLMs","venue":"cs.CL","work_id":"56766c95-7f7b-4db7-8563-f6df210ecdd1","year":2024},"citing_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-16T14:26:38.517787Z"},"links":{"cited_paper":"/paper/2412.18925","citing_paper":"/paper/2501.18362"},"observation_digest":"sha256:da897ab4b6fec27472361c1884424311f3e181816d4f7645ea530327b6c6ae23","observation_id":"bb8ecff3-5e41-4fc1-9a04-78ece23abc7d","resolution":{"observed_at":"2026-05-16T14:26:38.534771Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The reduction was successful, as indicated by follow-up x-rays","venue":null,"work_id":"d278541d-2d93-4003-9fb5-05601831f070","year":null},"citing_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-16T14:26:38.517787Z"},"links":{"citing_paper":"/paper/2501.18362"},"observation_digest":"sha256:3d8c7f52c47cb7a9fa855dc89e2af135c1ee2d3eefc4b6e951829613e43d046d","observation_id":"144bd2dc-e05d-46d3-b2fb-19978cdddf0d","resolution":{"observed_at":"2026-05-16T14:26:38.537822Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"7a820355-3386-4e2e-bf30-dfd8a852845c","year":null},"citing_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-16T14:26:38.517787Z"},"links":{"citing_paper":"/paper/2501.18362"},"observation_digest":"sha256:169454f55c90c3bea7f10c4267604a3d9382320444aacbef2df6a3e4e9734d71","observation_id":"a24f93cd-838b-4b83-a028-bda24f51f104","resolution":{"observed_at":"2026-05-16T14:26:38.540385Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"c69bed33-c83b-4fc1-bb04-bab743498255","year":null},"citing_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-16T14:26:38.517787Z"},"links":{"citing_paper":"/paper/2501.18362"},"observation_digest":"sha256:4515ae7fef69cf0f021c603bb7311c37a9cfbb89e96119dc88feff156b048f0a","observation_id":"a394bcdb-a3e7-44e2-a79c-3d95aba52d98","resolution":{"observed_at":"2026-05-16T14:26:38.542719Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"If there is a suspicion of nerve injury, such as the axillary nerve in this case, an EMG would be helpful in confirming nerve dysfunction or damage","venue":null,"work_id":"ec3f5c78-2daf-4147-90f5-edcca7a41643","year":null},"citing_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-16T14:26:38.517787Z"},"links":{"citing_paper":"/paper/2501.18362"},"observation_digest":"sha256:e7b6cd795b79dcd85ef61fa917d85fb078c20f4bafa8e4cc1c45da5413f5ec27","observation_id":"494336fe-6a00-456f-9a4c-b004e1310925","resolution":{"observed_at":"2026-05-16T14:26:38.545499Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"8ad7cf16-e2e9-4348-9e01-f4b4d6245233","year":null},"citing_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-16T14:26:38.517787Z"},"links":{"citing_paper":"/paper/2501.18362"},"observation_digest":"sha256:3762688fe0646eaefdd16fc8eadd592ed385e445e0e96e012f9ae057b50d5d0f","observation_id":"13fc9d3c-116e-406d-888b-99931e708187","resolution":{"observed_at":"2026-05-16T14:26:38.547874Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"There are no distinct P waves visible before each QRS complex; instead, there is a disorganized electrical activity, which is typical of fibrillatory waves","venue":null,"work_id":"a0ef369b-1ab0-4180-b4e7-cf3cef562fb4","year":null},"citing_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-16T14:26:38.517787Z"},"links":{"citing_paper":"/paper/2501.18362"},"observation_digest":"sha256:09c7749883a931237ac92b7f561fecfe02f828c3286f62448963ad9b05560368","observation_id":"7b2860e3-3c64-417b-8ed4-e780e609a92d","resolution":{"observed_at":"2026-05-16T14:26:38.550655Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"saw-tooth","venue":null,"work_id":"f96594d1-a540-4861-b24f-55928e9b336e","year":null},"citing_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-16T14:26:38.517787Z"},"links":{"citing_paper":"/paper/2501.18362"},"observation_digest":"sha256:b1acb786e8f210ffca6ffe40320a560f6319a2a93c6a7bec0fe331f6a87d073e","observation_id":"3facc0df-cced-48ee-a493-8a4a4dbfe47b","resolution":{"observed_at":"2026-05-16T14:26:38.552858Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"saw-tooth","venue":null,"work_id":"08c9da8f-972e-4e57-94cd-ee644ff7dc77","year":null},"citing_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-16T14:26:38.517787Z"},"links":{"citing_paper":"/paper/2501.18362"},"observation_digest":"sha256:0a1e173eeda6d720d556e3ce50e835a9e6b34c39ea9dcea7357e8a1506970775","observation_id":"c1cf6fc9-dec9-49c2-91df-4edcf2cf25c9","resolution":{"observed_at":"2026-05-16T14:26:38.555139Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Analysis: The above response fails to fully grasp the question’s implications in several ways","venue":null,"work_id":"2319325c-2ec5-4864-8899-2eca566710f7","year":null},"citing_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-16T14:26:38.517787Z"},"links":{"citing_paper":"/paper/2501.18362"},"observation_digest":"sha256:c26a90f8438e8fa53f0e325d3262270df857667839eed6f246c7af76b4854f28","observation_id":"e134e893-1be6-4dbe-b30a-5c5e0436373d","resolution":{"observed_at":"2026-05-16T14:26:38.557551Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Aging is associated with alterations in sleep architecture, which may lead to reductions in total sleep duration, slow-wave (deep) sleep, and sleep efficiency","venue":null,"work_id":"19057024-bf2c-4bae-86b2-14690357e877","year":null},"citing_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-16T14:26:38.517787Z"},"links":{"citing_paper":"/paper/2501.18362"},"observation_digest":"sha256:e40301a3d4a96a1dfa83de4cdda8c4e875aea767708b8efba9aac5679640cefd","observation_id":"91e08cf8-73d3-4bf2-8f08-c83aa91de78b","resolution":{"observed_at":"2026-05-16T14:26:38.560183Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"His sleep appears peaceful, and he experiences no disruptive symptoms such as snoring or awakenings","venue":null,"work_id":"11e67072-2c2e-433e-a49c-714dda5eadf5","year":null},"citing_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-16T14:26:38.517787Z"},"links":{"citing_paper":"/paper/2501.18362"},"observation_digest":"sha256:84fa927d4a292c85faf19528e8ccf3a895cbd26ee7b7aad71aa5f578106ba840","observation_id":"44bc7d9a-0b50-45ea-843f-7d49e3d7fae5","resolution":{"observed_at":"2026-05-16T14:26:38.562466Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"42fda939-1667-4f1c-aa97-a9ed539dd35b","year":null},"citing_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-16T14:26:38.517787Z"},"links":{"citing_paper":"/paper/2501.18362"},"observation_digest":"sha256:a987294ae21522fdacb60722ae1569fe355946c7f411d5dfc513d3e5d78ff66a","observation_id":"03263f01-f0ef-406f-8075-78a8ca491727","resolution":{"observed_at":"2026-05-16T14:26:38.564519Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"This can result in less restorative sleep and a less refreshed feeling upon waking","venue":null,"work_id":"5d4345a1-630a-4879-98b9-8a687b22c16d","year":null},"citing_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-16T14:26:38.517787Z"},"links":{"citing_paper":"/paper/2501.18362"},"observation_digest":"sha256:1602c1ed6c3d2929508f304d18f1de3b15002eb4b1ad810f1a236c7b02711223","observation_id":"048946b8-4d64-4023-8398-86d91d47169a","resolution":{"observed_at":"2026-05-16T14:26:38.566963Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"acute on chronic","venue":null,"work_id":"5e449c09-f433-4ebb-b303-a09941003ecb","year":null},"citing_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-16T14:26:38.517787Z"},"links":{"citing_paper":"/paper/2501.18362"},"observation_digest":"sha256:12d27b41fca9d11a65114b5a6e648628892a2a2b20a6282d7899dd3c9c9580c4","observation_id":"af6f1089-91f7-4814-98ef-0fac92b684fa","resolution":{"observed_at":"2026-05-16T14:26:38.569119Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"9e6091eb-dba0-45d1-b78f-9e455c73c4af","year":null},"citing_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-16T14:26:38.517787Z"},"links":{"citing_paper":"/paper/2501.18362"},"observation_digest":"sha256:0bc6ced1e7b0788b473cd8d8e0fc3122ab6e4e4145e27383485e66f72875fa7f","observation_id":"0b7fce81-589c-4ef7-98af-3934cf197471","resolution":{"observed_at":"2026-05-16T14:26:38.571248Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"13a642ea-638d-49d6-a524-81f93e7ea323","year":null},"citing_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-16T14:26:38.517787Z"},"links":{"citing_paper":"/paper/2501.18362"},"observation_digest":"sha256:6c15a81234708b487781c6a11fac29d99003b936c77c791c120637eb75c10c88","observation_id":"63adbc35-7d0f-4985-af8d-914ce78c9627","resolution":{"observed_at":"2026-05-16T14:26:38.573509Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"0a6051cc-f9b6-464e-9218-3cef9b6f9e5b","year":null},"citing_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-16T14:26:38.517787Z"},"links":{"citing_paper":"/paper/2501.18362"},"observation_digest":"sha256:c41a456c5a3603bc2a7dc9e20d0ffd5a57a4750f6f4b268484581e2e616dcb06","observation_id":"1ad086e1-5e90-4a2e-a789-8a6433785b81","resolution":{"observed_at":"2026-05-16T14:26:38.575528Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Rigorously ensure clarity and avoid ambiguity","venue":null,"work_id":"fc1a0ada-0acd-4642-aebd-92cd954c3228","year":null},"citing_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-16T14:26:38.517787Z"},"links":{"citing_paper":"/paper/2501.18362"},"observation_digest":"sha256:6f0f2774e81651bf4203df3e2b89f412b053eefd40176172aeeba06946e1aa2e","observation_id":"7e3fced3-a82e-42d0-b134-291150473ca8","resolution":{"observed_at":"2026-05-16T14:26:38.577739Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Do not change, add, or delete any factual information","venue":null,"work_id":"6210f480-83ff-429a-a526-65210c534395","year":null},"citing_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-16T14:26:38.517787Z"},"links":{"citing_paper":"/paper/2501.18362"},"observation_digest":"sha256:3a5d7b9bd3852d416d32e800077e668dd651045cafb3dff3d9f198920df507e3","observation_id":"7856cec6-a23f-4446-bc6d-471b0a40c916","resolution":{"observed_at":"2026-05-16T14:26:38.579861Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Pay special attention to keep any tabular data in completely the same format as the original","venue":null,"work_id":"5985e9db-093c-4cbd-b35c-e5056c80c29f","year":null},"citing_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-16T14:26:38.517787Z"},"links":{"citing_paper":"/paper/2501.18362"},"observation_digest":"sha256:36bc1ba96f66045a02bcc476f7e7f7a4d3f4470930ef73c1849c83787e63c0ec","observation_id":"a5e981b8-a5e7-413e-aa37-c69650314810","resolution":{"observed_at":"2026-05-16T14:26:38.582069Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Answer Choices: (A) [Option A] (B) [Option B]","venue":null,"work_id":"3f0d098b-1f02-48f8-9727-e3e9bd49e7c7","year":null},"citing_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-16T14:26:38.517787Z"},"links":{"citing_paper":"/paper/2501.18362"},"observation_digest":"sha256:2a0b1058b86566000eab2a7ed028afc054296ccec1bc256edff429b47f365672","observation_id":"04249930-be57-43ac-90f2-a29cc6aeb007","resolution":{"observed_at":"2026-05-16T14:26:38.584097Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"53f17873-de27-402a-9dfc-cb0626546aac","year":null},"citing_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-16T14:26:38.517787Z"},"links":{"citing_paper":"/paper/2501.18362"},"observation_digest":"sha256:0dba618f58d2411dcfbd38371e912b2ef41c5947ed7ed6bb1a43740ed14b46ed","observation_id":"c0c44156-0ac3-4f5f-b984-0c339de9d2d4","resolution":{"observed_at":"2026-05-16T14:26:38.587307Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"b67865cd-e2c1-45da-a573-49d678ee5888","year":null},"citing_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-16T14:26:38.517787Z"},"links":{"citing_paper":"/paper/2501.18362"},"observation_digest":"sha256:8120c20ca799106a0fc9f1c2cc119d50a426603c0eb6799f0b30563a318f8052","observation_id":"dc22a6d1-3331-4cd4-a04c-14de87b7b8d7","resolution":{"observed_at":"2026-05-16T14:26:38.589630Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"8bf624b4-2ab0-4fd6-8529-1e7ef5a84b7a","year":null},"citing_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-16T14:26:38.517787Z"},"links":{"citing_paper":"/paper/2501.18362"},"observation_digest":"sha256:7b6f4140563f2d201d1aeb115d52909248bbeb0c41b7b589a28627ed5122ed70","observation_id":"8eaa5b77-a94d-4497-ad15-9f09edad1f54","resolution":{"observed_at":"2026-05-16T14:26:38.591786Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"They should be clear, concise, and professionally worded","venue":null,"work_id":"e51189c1-4ffb-4d94-bdd6-f28b4b11d84b","year":null},"citing_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-16T14:26:38.517787Z"},"links":{"citing_paper":"/paper/2501.18362"},"observation_digest":"sha256:ea27af6f8b032764e3406c0fc27630e2c26b733e53e08ecfd64e005806bf6ec0","observation_id":"33df4d60-d8a3-49d9-8873-34883d712933","resolution":{"observed_at":"2026-05-16T14:26:38.594021Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"65797011-f44e-4aa1-be51-7d594e736d94","year":null},"citing_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-16T14:26:38.517787Z"},"links":{"citing_paper":"/paper/2501.18362"},"observation_digest":"sha256:0d838fc045b4d9e4549e82a5ce4b4e12a3de07d1c494963c16cc68ec31ba3249","observation_id":"4ce60eb3-9ac6-41ff-b88f-cfd712f2b10c","resolution":{"observed_at":"2026-05-16T14:26:38.596171Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Avoid options that are overtly illogical or unsupported.,→","venue":null,"work_id":"bbac3e1a-c0f8-405d-8f7e-d900edb562ee","year":null},"citing_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-16T14:26:38.517787Z"},"links":{"citing_paper":"/paper/2501.18362"},"observation_digest":"sha256:1ec30e5520ddd9a88ac10e66fe422bff83b3ac278d8c66805c93a64d8233234e","observation_id":"e5cd1a33-ff7a-433f-8547-1c28770ffc76","resolution":{"observed_at":"2026-05-16T14:26:38.598395Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Answer Choices: (A) [Option A] (B) [Option B]","venue":null,"work_id":"57f8443f-e4da-418d-aae0-e1ee39ca1036","year":null},"citing_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-16T14:26:38.517787Z"},"links":{"citing_paper":"/paper/2501.18362"},"observation_digest":"sha256:6bcb2fa9a25708f311a40791fff70341ed68305127c894894b84c6429ee06348","observation_id":"135efca3-6d51-4b18-bfe8-1f01ca87ed36","resolution":{"observed_at":"2026-05-16T14:26:38.602926Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","latest_version":3,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-17T19:58:59.903876Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding"},"reference_resolution":{"displayed":30,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":11,"verified_exact":0,"verified_fuzzy":18},"total_outbound_references":30},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"thesis":"As of 23 August 2026, this Paper Citation Record lists 30 of 30 outbound references and 88 inbound Pith citation observations for arXiv:2501.18362."}