{"as_of":"2026-08-07T11:53:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:de9b455e6159cf5fb1bd69f1e118cad13b58869f63ea69e69f01f2d13dd3ccb0","coverage":[{"denominator":192,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T11:45:41.260152Z","state":"measured"},{"denominator":105,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":105,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":5,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":5,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-01T14:15:55.452066Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-25T04:55:23.438961Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"cited_work":{"arxiv_id":"2508.00923","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2508.00923","snapshot_observed_at":"2026-07-17T01:20:37.408702Z","title":"Beyond benchmarks: Dynamic, automatic and systematic red-teaming agents for trustworthy medical language models","venue":null,"work_id":"768b946f-4a7a-47de-ab43-21518e080703","year":2025},"citing_paper":{"arxiv_id":"2604.14475","last_updated":"2026-04-15T23:12:02Z","snapshot_observed_at":"2026-07-06T23:02:13.885476Z","submitted_at":"2026-04-15T23:12:02Z","title":"Evo-MedAgent: Beyond One-Shot Diagnosis with Agents That Remember, Reflect, and Improve","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T12:35:32.915685Z"},"links":{"cited_paper":"/paper/2508.00923","citing_paper":"/paper/2604.14475"},"observation_digest":"sha256:d96e223dc47253d384d110dab4124dd3368d43b65af29d05cf62c27576fa60d5","observation_id":"07db95af-d619-4d98-97b1-1b025bc2ce47","resolution":{"observed_at":"2026-07-17T01:20:37.408702Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"cited_work":{"arxiv_id":"2508.00923","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2508.00923","snapshot_observed_at":"2026-07-17T01:20:37.408702Z","title":"Beyond benchmarks: Dynamic, automatic and systematic red-teaming agents for trustworthy medical language models","venue":null,"work_id":"768b946f-4a7a-47de-ab43-21518e080703","year":2025},"citing_paper":{"arxiv_id":"2604.26959","last_updated":"2026-04-07T10:54:57Z","snapshot_observed_at":"2026-07-06T23:12:29.385105Z","submitted_at":"2026-04-07T10:54:57Z","title":"CareGuardAI: Context-Aware Multi-Agent Guardrails for Clinical Safety & Hallucination Mitigation in Patient-Facing LLMs","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-10T19:17:27.351727Z"},"links":{"cited_paper":"/paper/2508.00923","citing_paper":"/paper/2604.26959"},"observation_digest":"sha256:c986fac80d328bb162333389333329c87e367bcd43a6a81b37c50f668220c6c9","observation_id":"cda51f17-a359-42f1-a408-55c7d11377bf","resolution":{"observed_at":"2026-07-17T01:20:37.408702Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"cited_work":{"arxiv_id":"2508.00923","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2508.00923","snapshot_observed_at":"2026-07-17T01:20:37.408702Z","title":"Beyond benchmarks: Dynamic, automatic and systematic red-teaming agents for trustworthy medical language models","venue":null,"work_id":"768b946f-4a7a-47de-ab43-21518e080703","year":2025},"citing_paper":{"arxiv_id":"2605.01970","last_updated":"2026-05-15T06:42:15Z","snapshot_observed_at":"2026-07-06T23:15:07.159340Z","submitted_at":"2026-05-03T17:07:20Z","title":"Trojan Hippo: Weaponizing Agent Memory for Data Exfiltration","version":3},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-19T17:30:22.481943Z"},"links":{"cited_paper":"/paper/2508.00923","citing_paper":"/paper/2605.01970"},"observation_digest":"sha256:52f0f2815562d604cc6f0671eccaf85c647470dea136bc4c1a7cdd8e08c10e02","observation_id":"7a3131d4-742d-4251-ad7f-1d62f9f963d8","resolution":{"observed_at":"2026-07-17T01:20:37.408702Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"cited_work":{"arxiv_id":"2508.00923","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2508.00923","snapshot_observed_at":"2026-07-17T01:20:37.408702Z","title":"Beyond benchmarks: Dynamic, automatic and systematic red-teaming agents for trustworthy medical language models","venue":null,"work_id":"768b946f-4a7a-47de-ab43-21518e080703","year":2025},"citing_paper":{"arxiv_id":"2605.23629","last_updated":"2026-05-22T13:41:10Z","snapshot_observed_at":"2026-07-06T23:33:49.013121Z","submitted_at":"2026-05-22T13:41:10Z","title":"DDX-TRACE: A Benchmark for Medical Diagnostic Trajectories in VLMs","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-25T04:53:33.343509Z"},"links":{"cited_paper":"/paper/2508.00923","citing_paper":"/paper/2605.23629"},"observation_digest":"sha256:662cdd3eeddfa82b08b72c901e4401fca29153afbb0a5e50231d6a8ae3256ded","observation_id":"1332dd5e-e265-49ae-8761-9804f003af9b","resolution":{"observed_at":"2026-07-17T01:20:37.408702Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.00923","snapshot_observed_at":"2026-08-01T14:15:55.452066Z","title":"(Preprint arXiv:2508.00923.)","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.18828","last_updated":"2026-07-21T08:05:35Z","snapshot_observed_at":"2026-08-06T11:57:32.648129Z","submitted_at":"2026-07-21T08:05:35Z","title":"Evaluating medical AI under missing information: same-provider judges and human raters change apparent safety","version":1},"reference_index":2026,"source":"pdf_text","source_observed_at":"2026-08-01T14:15:55.452066Z"},"links":{"cited_paper":"/paper/2508.00923","citing_paper":"/paper/2607.18828"},"observation_digest":"sha256:997767d16873878fc43a1dd192dfab64bd6307983eb808128bac034d983b1e38","observation_id":"7cf89c43-e38a-4b34-9bb0-dc2231a5a174","resolution":{"observed_at":"2026-08-01T14:15:55.452066Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2508.00923/citation-record","integrity":"/paper/2508.00923/integrity","json":"/paper/2508.00923/citation-record.json","paper":"/paper/2508.00923"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:40.900266Z","title":"Toward expert-level medical question answering with large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:40.900266Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:c1100af19fd4e90d952314f34e6289d5ac59fd20411eb118d0e62d643d3e3439","observation_id":"12365366-421b-4b8e-9927-970a8979d41a","resolution":{"observed_at":"2026-08-06T11:45:40.900266Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.18416","last_updated":"2024-05-01T17:12:10Z","snapshot_observed_at":"2026-07-06T18:06:49.190172Z","submitted_at":"2024-04-29T04:11:28Z","title":"Capabilities of Gemini Models in Medicine","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.18416","snapshot_observed_at":"2026-08-06T11:45:40.905924Z","title":"Capabilities of gemini models in medicine","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:40.905924Z"},"links":{"cited_paper":"/paper/2404.18416","citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:67ff1e34e5cf5be6aec0495d66bbedb2b7adb2544502d5eb4c5e66b074dcdc8e","observation_id":"9e884f9a-cde5-4e98-83ed-f9a2e53e5867","resolution":{"observed_at":"2026-08-06T11:45:40.905924Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:40.910585Z","title":"Openai o3 and o4-mini system card","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:40.910585Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:2e67f22eedaa2796fc9d076ddb8440ff4f52de65a82f170dc712a42bad26260c","observation_id":"ad19c8d3-84e8-4e39-9240-f50f82954f99","resolution":{"observed_at":"2026-08-06T11:45:40.910585Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:40.914243Z","title":"What disease does this patient have? a large-scale open domain question answering dataset from medical exams","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:40.914243Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:234761077d500e1296fc4cc3743e2ed5a583b8348ae137d0ee5acd68abb4bc8c","observation_id":"15f4f758-d274-4815-8302-d4192660e773","resolution":{"observed_at":"2026-08-06T11:45:40.914243Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:40.919022Z","title":"Towards accurate differential diagnosis with large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:40.919022Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:07ba757d03a941c50b1e1ec38bae85ab5e0223e7569177f81e3e92b053ce90fd","observation_id":"5516a013-fb85-4bc4-8da8-540b6aa3cd24","resolution":{"observed_at":"2026-08-06T11:45:40.919022Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:40.922690Z","title":"Feasibility of differential diagnosis based on imaging patterns using a large language model","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:40.922690Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:817c776ef72f3c9c2fe0e58235391e570875d3a99cc84045fe9dcbbfd88e4771","observation_id":"36d129d2-7070-450c-a76a-12d8661e970f","resolution":{"observed_at":"2026-08-06T11:45:40.922690Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:40.926556Z","title":"Mdagents: An adaptive collaboration of llms for medical decision-making","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:40.926556Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:0cc103aa3d7e0695abe28e93ae448e4e70fcc4130f517a6d695d34e3700bbfe2","observation_id":"69519177-5f6c-4169-be88-66a2c33c21a9","resolution":{"observed_at":"2026-08-06T11:45:40.926556Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:40.930364Z","title":"Chatgpt as a tool for medical education and clinical decision-making on the wards: case study","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:40.930364Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:47388db4cece06d8e7d62edb9a9b51596d0b15f283ac9a5c2a99102fcc4e7576","observation_id":"1bba5ec7-eb5a-470e-91b8-1f91a2685a26","resolution":{"observed_at":"2026-08-06T11:45:40.930364Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:40.934935Z","title":"Food and Drug Administration","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:40.934935Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:7119a6c5439c3a00e7ed78ea5bcd123379d849e193ec70643a6c498867293e68","observation_id":"2c1d038d-c523-471b-af96-4dadd8960009","resolution":{"observed_at":"2026-08-06T11:45:40.934935Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:40.938486Z","title":"Problems of monetary management: the UK experience","venue":null,"work_id":null,"year":1984},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:40.938486Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:6dca6def940a8828e88010cb272edd4b16c137573a26757650811fd826689b44","observation_id":"8bf70763-be16-41dc-8f00-786e4d8b4df8","resolution":{"observed_at":"2026-08-06T11:45:40.938486Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:40.941788Z","title":"Jailbreaking black box large language models in twenty queries","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:40.941788Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:14bacecc9d2bb77b63f9c5f795ee2cec12ca90216e5c6d0962cc9278dcfda6c1","observation_id":"4b842c9e-c1ac-4f97-9375-60ebb90f5973","resolution":{"observed_at":"2026-08-06T11:45:40.941788Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:40.945757Z","title":"Rainbow teaming: Open-ended generation of diverse adversarial prompts","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:40.945757Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:7cfeb86440146f2a2c34d173464167fbcc01a6a0c48f47f7804780ede0217447","observation_id":"50761e78-651b-4036-a1b8-ca416b1d0e70","resolution":{"observed_at":"2026-08-06T11:45:40.945757Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05295","last_updated":"2025-04-22T05:20:39Z","snapshot_observed_at":"2026-07-06T19:29:14.335016Z","submitted_at":"2024-10-03T17:59:01Z","title":"AutoDAN-Turbo: A Lifelong Agent for Strategy Self-Exploration to Jailbreak LLMs","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05295","snapshot_observed_at":"2026-08-06T11:45:40.949098Z","title":"Autodan-turbo: A lifelong agent for strategy self-exploration to jailbreak llms","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:40.949098Z"},"links":{"cited_paper":"/paper/2410.05295","citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:b0ebcc249cbc27f6baa83b422be8d927bde615686a7374c6c9e3657ab9af1d82","observation_id":"e7d2450e-b32f-4a61-acd8-d6f99824c51d","resolution":{"observed_at":"2026-08-06T11:45:40.949098Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:40.953105Z","title":"Wildteaming at scale: From in-the-wild jailbreaks to (adversarially) safer language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:40.953105Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:f59228a4a295dabd37ea436867554d7a0398810f3e6b3191491bafe196a05ca7","observation_id":"b30780c6-ed28-4cb8-9a57-43e8ca71fdb7","resolution":{"observed_at":"2026-08-06T11:45:40.953105Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:40.956595Z","title":"Red teaming chatgpt in medicine to yield real-world insights on model behavior","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:40.956595Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:3ad91ef55b023b4b34fd637d0d202b30af8f6baadd6e787279db991c027babcc","observation_id":"c38ff15f-a527-4720-98d2-a8f88a129872","resolution":{"observed_at":"2026-08-06T11:45:40.956595Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2507.07248","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:42.233883Z","title":"Medical red teaming protocol of language models: On the importance of user perspectives in healthcare settings","venue":null,"work_id":"011dbcbd-4560-4f4e-9102-c2ec4d2f99e3","year":2025},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:40.960138Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:9164938a33134db4f765fb6dbfa0b6ab4b5f29881342802018bf5408c442a777","observation_id":"78abb78a-3f6f-4835-a78e-25bb2946daab","resolution":{"observed_at":"2026-08-06T11:45:42.240853Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:40.963903Z","title":"Medmcqa: A large-scale multi-subject multi-choice dataset for medical domain question answering","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:40.963903Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:1cce74637316788e217d32a75cf16c7091e7b959347032aa3e7b77f914b23733","observation_id":"63b5faa8-4f8f-421d-a2db-58300344fcab","resolution":{"observed_at":"2026-08-06T11:45:40.963903Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:40.968359Z","title":"Evaluation and mitigation of the limitations of large language models in clinical decision-making","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:40.968359Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:f9145f14a0e62c5a1c9b46bacb0620fced948c22c6c3a03852e3881f3e304524","observation_id":"9fae5d2b-4c08-4bd4-ab1b-fda892db326d","resolution":{"observed_at":"2026-08-06T11:45:40.968359Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18362","last_updated":"2025-06-06T13:00:07Z","snapshot_observed_at":"2026-07-06T20:28:24.203633Z","submitted_at":"2025-01-30T14:07:56Z","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18362","snapshot_observed_at":"2026-08-06T11:45:40.971770Z","title":"Medxpertqa: Benchmarking expert-level medical reasoning and understanding","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:40.971770Z"},"links":{"cited_paper":"/paper/2501.18362","citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:e319d3a9881f2ce4eebb3a0c8ee95977a1ad80293033f6526f988f1dd9312485","observation_id":"0c6493ec-dbf9-4033-8a9e-b6e045fa6b1d","resolution":{"observed_at":"2026-08-06T11:45:40.971770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.07459","last_updated":"2026-06-16T17:07:03Z","snapshot_observed_at":"2026-07-06T20:50:00.422560Z","submitted_at":"2025-03-10T15:38:44Z","title":"MedicalAgentsBench for Complex Medical Reasoning: Comparing Internalized Reasoning Models versus Externalized Agent-based Frameworks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.07459","snapshot_observed_at":"2026-08-06T11:45:40.975815Z","title":"Medagentsbench: Benchmarking thinking models and agent frameworks for complex medical reasoning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:40.975815Z"},"links":{"cited_paper":"/paper/2503.07459","citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:78a408bb353b05a3afff3e3f5f71f4d75538cbfa77a740602a1c24e9b568520f","observation_id":"f0d8750a-1f3e-48e7-8dfa-0090bf67107c","resolution":{"observed_at":"2026-08-06T11:45:40.975815Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.00467","last_updated":"2025-07-11T14:39:47Z","snapshot_observed_at":"2026-08-06T08:52:58.857670Z","submitted_at":"2025-05-01T11:43:27Z","title":"Red Teaming Large Language Models for Healthcare","version":2},"cited_work":{"arxiv_id":"2505.00467","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.00467","snapshot_observed_at":"2026-08-06T11:45:42.143254Z","title":"Red Teaming Large Language Models for Healthcare","venue":"cs.CL","work_id":"5a7a8b49-8a72-4718-89df-bdb65d65cec0","year":2025},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:40.979660Z"},"links":{"cited_paper":"/paper/2505.00467","citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:231057a64f0c1cb20d650601feaa88ddcffd5034d83b2870e3ec1ab53a0038de","observation_id":"71f78e35-7557-4c06-9c06-0823df3c1706","resolution":{"observed_at":"2026-08-06T11:45:42.146944Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:40.983171Z","title":"Large language models propagate race-based medicine","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:40.983171Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:b97f9673d898531828ed1a64c5cf6354a2ef9df954df05ff0e29f56976ad65f7","observation_id":"4a55690a-d211-425c-88dd-d262cc96918e","resolution":{"observed_at":"2026-08-06T11:45:40.983171Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:40.986532Z","title":"A framework to assess clinical safety and hallucination rates of llms for medical text summarisation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:40.986532Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:04a38fc720e52178e6520a55b6e8a5439176efa4fda730ed13aba4b0dea8e4b5","observation_id":"fc864a6d-7a5e-4674-81c5-a663cdcacd49","resolution":{"observed_at":"2026-08-06T11:45:40.986532Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:40.989806Z","title":"A toolbox for surfacing health equity harms and biases in large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:40.989806Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:f8f5c665685bf5ec23042700e4e351ce46c3e2c47951260c60c6b6057d71296a","observation_id":"15cce580-1ba2-4263-b15d-029e4dd23433","resolution":{"observed_at":"2026-08-06T11:45:40.989806Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:40.993016Z","title":"Amqa: An adversarial dataset for benchmarking bias of llms in medicine and healthcare","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:40.993016Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:739fe8e1b934afd5495610c8ae26f5b5ea3db9befcd6184bce3c344fe7137150","observation_id":"5d82aeaa-08d8-4e11-a671-a076b319df13","resolution":{"observed_at":"2026-08-06T11:45:40.993016Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:40.996536Z","title":"Evaluation and mitigation of cognitive biases in medical language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:40.996536Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:4b28c730c812b2a5b93b644405b910d161e57aa63c6364e2d4e781cd292ddbc2","observation_id":"a9f659e5-4433-4b97-a8f6-35852531c3a7","resolution":{"observed_at":"2026-08-06T11:45:40.996536Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.02730","last_updated":"2024-07-03T00:59:03Z","snapshot_observed_at":"2026-07-06T18:40:39.990816Z","submitted_at":"2024-07-03T00:59:03Z","title":"MedVH: Towards Systematic Evaluation of Hallucination for Large Vision Language Models in the Medical Context","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.02730","snapshot_observed_at":"2026-08-06T11:45:40.999677Z","title":"Medvh: Towards systematic evaluation of hallucination for large vision language models in the medical context","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:40.999677Z"},"links":{"cited_paper":"/paper/2407.02730","citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:d57c5c207b7c3ad0e35cc328ada0716a9bdfa06593c4f7a4905a6c7830bf5b54","observation_id":"ecc4e126-c511-4fba-99d3-4fe948daac19","resolution":{"observed_at":"2026-08-06T11:45:40.999677Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.01201","last_updated":"2025-04-01T21:34:01Z","snapshot_observed_at":"2026-07-06T21:02:37.952630Z","submitted_at":"2025-04-01T21:34:01Z","title":"Medical large language models are easily distracted","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.01201","snapshot_observed_at":"2026-08-06T11:45:41.003724Z","title":"Medical large language models are easily distracted","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.003724Z"},"links":{"cited_paper":"/paper/2504.01201","citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:3f94f5139922270516e795e8627ca230c102c004ec81bd5df83c622881c6ec62","observation_id":"80b8d0b9-76d2-41e2-878b-c5812050bc7d","resolution":{"observed_at":"2026-08-06T11:45:41.003724Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.007619Z","title":"Last updated 12 Jul 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.007619Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:60d6462243f2c09e4755fe1a18c8869f552f8d446875c4912946918b6b88b570","observation_id":"fe248e7d-62d6-470a-bbf1-0a23e1cf4141","resolution":{"observed_at":"2026-08-06T11:45:41.007619Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.011057Z","title":"The 10 most common hipaa violations you should avoid","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.011057Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:0df6b17db5b174983136b086cdc49541577dd22a55782a09b3e996db7d439a0a","observation_id":"8033be0c-b6f6-4c50-8f4b-762eaf4a3ffd","resolution":{"observed_at":"2026-08-06T11:45:41.011057Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.014511Z","title":"Accidental hipaa violation: Examples & how to respond effectively in 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.014511Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:edefd16f92eac9c5597c0aae0c5ed3780fe7d7f6b8bedc7e2123f2a26c6a8dfe","observation_id":"7ba4ca29-60d3-44c2-bc4d-9547010ae717","resolution":{"observed_at":"2026-08-06T11:45:41.014511Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.017727Z","title":"What is the proper response to an accidental hipaa violation? https://www.hipaaguide.net/ proper-response-to-an-accidental-hipaa-violation/ , 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.017727Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:249fca94dcc921ddd2380461eb3843e30ca8d31ac74a83aac677070f32a432b2","observation_id":"5dc66e95-05de-46cd-96d8-abad6f238d48","resolution":{"observed_at":"2026-08-06T11:45:41.017727Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.020800Z","title":"Could human error cause a data breach un- der the gdpr? https://www.privacycompliancehub.com/gdpr-resources/ could-human-error-cause-a-data-breach-under-the-gdpr/ , 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.020800Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:313af5c92bd7efccbd2b5871d30ab69337468332c5651ddf2f68d196ad6b729e","observation_id":"940ab202-c43d-4551-a3c5-32d8320ce76f","resolution":{"observed_at":"2026-08-06T11:45:41.020800Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.024348Z","title":"Sociodemographic biases in medical decision making by large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.024348Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:a9edb45af8f3fd2cfed805d444540eb7d4a96ff2467578618dfa468e6d116a72","observation_id":"c9c55ddf-f130-4162-9057-cb729e56c2fa","resolution":{"observed_at":"2026-08-06T11:45:41.024348Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.08775","last_updated":"2025-05-13T17:53:59Z","snapshot_observed_at":"2026-07-06T21:23:23.760393Z","submitted_at":"2025-05-13T17:53:59Z","title":"HealthBench: Evaluating Large Language Models Towards Improved Human Health","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.08775","snapshot_observed_at":"2026-08-06T11:45:41.027391Z","title":"Healthbench: Evaluating large language models towards improved human health","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.027391Z"},"links":{"cited_paper":"/paper/2505.08775","citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:2cc09f205178143910cdcac1554bb3cff725e193476b1fa6513e43aba0994810","observation_id":"147a0e6c-3188-463c-b769-cab376dde6e7","resolution":{"observed_at":"2026-08-06T11:45:41.027391Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.030927Z","title":"A large language model for electronic health records","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.030927Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:bc9ff8bf84a5a637978d77fc4400f4db127d4c6c5ed94c356dd5d3a22d02c13f","observation_id":"9a9819ef-7a5a-440a-9c55-035190bf1504","resolution":{"observed_at":"2026-08-06T11:45:41.030927Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.034032Z","title":"Instruction tuning large language models to understand electronic health records","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.034032Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:eab50a1a7b14e139c8e8b89fbedf31e3c3e64710d1c07d8654fa463e423896b6","observation_id":"c7e5e6de-7dc9-499e-a19f-bb99cf8d4401","resolution":{"observed_at":"2026-08-06T11:45:41.034032Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.037171Z","title":"Large language models for chatbot health advice studies: a systematic review","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.037171Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:07835fe050ccb291e7ad425b3ae3d4e04064f167f91dd53a2480600e6cd2c93b","observation_id":"9bfd3b16-8678-425e-b8fd-2f03424d3a4c","resolution":{"observed_at":"2026-08-06T11:45:41.037171Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.040136Z","title":"Contextual integrity in llms via reasoning and reinforcement learning.arXiv preprint arXiv:2506.04245, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.040136Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:c106329002146599c9bc7e1c7a87d591ffa31149a90c84868adc95e7c871c6f3","observation_id":"92d5488e-1924-4490-af25-17c66b1f9008","resolution":{"observed_at":"2026-08-06T11:45:41.040136Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.09277","last_updated":"2023-11-15T18:54:01Z","snapshot_observed_at":"2026-08-02T16:03:14.942403Z","submitted_at":"2023-11-15T18:54:01Z","title":"Contrastive Chain-of-Thought Prompting","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.09277","snapshot_observed_at":"2026-08-06T11:45:41.043610Z","title":"Contrastive chain-of-thought prompting","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.043610Z"},"links":{"cited_paper":"/paper/2311.09277","citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:04ee231f2a5a7aaae8f96c9c5b7cb346817d290642dee579eaf7b1cec430ae19","observation_id":"fc6be592-762b-431f-a64e-49cc20dfcf22","resolution":{"observed_at":"2026-08-06T11:45:41.043610Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.01879","last_updated":"2023-08-30T21:28:01Z","snapshot_observed_at":"2026-08-06T01:14:28.656007Z","submitted_at":"2023-05-03T03:47:00Z","title":"SCOTT: Self-Consistent Chain-of-Thought Distillation","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.01879","snapshot_observed_at":"2026-08-06T11:45:41.047345Z","title":"Scott: Self-consistent chain-of-thought distillation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.047345Z"},"links":{"cited_paper":"/paper/2305.01879","citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:a5eea918073986158a1785c7f123d9e118bbbc0ba9953429760243e6a6b3b509","observation_id":"456fc642-84dc-4721-a096-ad3bfd2e710a","resolution":{"observed_at":"2026-08-06T11:45:41.047345Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.051317Z","title":"Llava-med: Training a large language-and-vision assistant for biomedicine in one day","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.051317Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:64ba011e379bfcba60f611bbc1601747a18d1d03927d4c30ff018ea1619d79db","observation_id":"4f5f4179-0e60-497e-870a-7e1ac45d47f5","resolution":{"observed_at":"2026-08-06T11:45:41.051317Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.054788Z","title":"A generalist vision–language foundation model for diverse biomedical tasks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.054788Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:edde66ae993afa4b0b190d6c0a63aebd833aa56c695b0b22cc2c133e15f47437","observation_id":"53ef1dc0-d1c8-409d-9a9c-04f774a82c02","resolution":{"observed_at":"2026-08-06T11:45:41.054788Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19634","last_updated":"2025-03-19T13:55:33Z","snapshot_observed_at":"2026-07-06T20:43:22.572496Z","submitted_at":"2025-02-26T23:57:34Z","title":"MedVLM-R1: Incentivizing Medical Reasoning Capability of Vision-Language Models (VLMs) via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.19634","snapshot_observed_at":"2026-08-06T11:45:41.058150Z","title":"Medvlm-r1: Incentivizing medical reasoning capability of vision-language models (vlms) via reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.058150Z"},"links":{"cited_paper":"/paper/2502.19634","citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:73047f68fdb797acc4a5edc0eba820da07874e589051b7d25d0c642288ad49a8","observation_id":"e06e925f-1684-4ceb-85d4-210433870bd8","resolution":{"observed_at":"2026-08-06T11:45:41.058150Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.061704Z","title":"A visual-language foundation model for computational pathology","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.061704Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:4ff0374b40360105eff25f365cf9c914beb817132d70031a4c4e1a9a2964ca89","observation_id":"98b7d4e2-9eca-4489-8eca-f2d746212d0c","resolution":{"observed_at":"2026-08-06T11:45:41.061704Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.17012","last_updated":"2024-09-25T16:57:20Z","snapshot_observed_at":"2026-07-06T16:25:21.571679Z","submitted_at":"2023-09-29T06:53:10Z","title":"Benchmarking Cognitive Biases in Large Language Models as Evaluators","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.17012","snapshot_observed_at":"2026-08-06T11:45:41.064778Z","title":"Benchmarking cognitive biases in large language models as evaluators","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.064778Z"},"links":{"cited_paper":"/paper/2309.17012","citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:2ec6cd28cf2e633c94121512e5c9a36ba3df6fbde125b7b12db5cf79b50fefe9","observation_id":"602d4ffe-c342-4bef-9e26-f28960e02120","resolution":{"observed_at":"2026-08-06T11:45:41.064778Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.01358","last_updated":"2023-08-28T20:10:55Z","snapshot_observed_at":"2026-07-06T15:11:39.641284Z","submitted_at":"2023-04-03T20:30:59Z","title":"Challenging the appearance of machine intelligence: Cognitive bias in LLMs and Best Practices for Adoption","version":3},"cited_work":{"arxiv_id":"2304.01358","doi":null,"metadata_source":"pith","pith_arxiv_id":"2304.01358","snapshot_observed_at":"2026-08-06T11:45:41.913612Z","title":"Challenging the appearance of machine intelligence: Cognitive bias in LLMs and Best Practices for Adoption","venue":"cs.HC","work_id":"a39c945a-cc3c-47fa-afaf-7780a651e21a","year":2023},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.069297Z"},"links":{"cited_paper":"/paper/2304.01358","citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:44f597728fde82203caa9cb67fd3ea4706386cfbc2d0e251508dab80dedb4680","observation_id":"ee997568-35c6-4268-a27f-fe8c7a2840ae","resolution":{"observed_at":"2026-08-06T11:45:41.918504Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.18070","last_updated":"2024-06-17T15:08:05Z","snapshot_observed_at":"2026-07-06T17:23:14.388964Z","submitted_at":"2024-01-31T18:48:20Z","title":"Do Language Models Exhibit the Same Cognitive Biases in Problem Solving as Human Learners?","version":2},"cited_work":{"arxiv_id":"2401.18070","doi":null,"metadata_source":"pith","pith_arxiv_id":"2401.18070","snapshot_observed_at":"2026-08-06T11:45:41.895796Z","title":"Do Language Models Exhibit the Same Cognitive Biases in Problem Solving as Human Learners?","venue":"cs.CL","work_id":"31a8ab16-7602-4b00-b558-01a19abed9c4","year":2024},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.073178Z"},"links":{"cited_paper":"/paper/2401.18070","citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:3874332a9032385441c142990237d7c22ff95241403abf85690f655974cc282f","observation_id":"84c6c729-6dbd-4120-960c-7ecfd0f45a3c","resolution":{"observed_at":"2026-08-06T11:45:41.900933Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.077129Z","title":"Cognitive bias in high-stakes decision-making with llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.077129Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:bdcf67a215a70d7e8704ff3f08bac8f8aca44d311c203a553d3e10d5f023b402","observation_id":"bc4ddc28-14c7-4b61-a5b6-ad855c45675f","resolution":{"observed_at":"2026-08-06T11:45:41.077129Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.02544","last_updated":"2025-09-05T09:21:10Z","snapshot_observed_at":"2026-08-05T10:11:53.289326Z","submitted_at":"2024-08-05T15:16:22Z","title":"Caution for the Environment: Multimodal LLM Agents are Susceptible to Environmental Distractions","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.02544","snapshot_observed_at":"2026-08-06T11:45:41.080721Z","title":"Caution for the environment: Multimodal agents are susceptible to environmental distractions","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.080721Z"},"links":{"cited_paper":"/paper/2408.02544","citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:4f69ee2e74b8634c9277908ba4553daa089cd040e2bc4299a7d69036ac069352","observation_id":"33e54c00-18ed-4414-96d2-28455256121d","resolution":{"observed_at":"2026-08-06T11:45:41.080721Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.084615Z","title":"Breaking focus: Contextual distraction curse in large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.084615Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:98b662ed3667cecc29504f925929288a594e749899ed9d4310f4e884d11da06b","observation_id":"abccac8d-f50c-48b5-8c49-c44446a21bd6","resolution":{"observed_at":"2026-08-06T11:45:41.084615Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.087690Z","title":"Distraction is all you need for multimodal large language model jailbreaking","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.087690Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:5c6a6c075b666999b6afdbb38269e1ab5d695d9d9242031406f499de9eb57d6c","observation_id":"2241e9ab-b378-445d-b93f-7901415964ba","resolution":{"observed_at":"2026-08-06T11:45:41.087690Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.04362","last_updated":"2025-02-05T04:52:57Z","snapshot_observed_at":"2026-07-06T20:32:23.502480Z","submitted_at":"2025-02-05T04:52:57Z","title":"LLMs can be easily Confused by Instructional Distractions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.04362","snapshot_observed_at":"2026-08-06T11:45:41.091522Z","title":"Llms can be easily confused by instructional distractions","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.091522Z"},"links":{"cited_paper":"/paper/2502.04362","citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:91c9accbb8a9630f4dda83541ad563a2472a371140c82301aff4b51bbe764504","observation_id":"b72f6d4b-c40e-48ce-a66f-36205a9b4458","resolution":{"observed_at":"2026-08-06T11:45:41.091522Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.094924Z","title":"Large language models are highly vulnerable to adversarial hallucination attacks in clinical decision support: A multi-model assurance analysis","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.094924Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:0596539eef9150cdd3f47ece4338f0529c67466d43f4a3c08304fed7b4a852fa","observation_id":"982cc3c0-1948-45fc-a5b8-cdbcc92004e5","resolution":{"observed_at":"2026-08-06T11:45:41.094924Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14302","last_updated":"2025-02-20T06:33:23Z","snapshot_observed_at":"2026-07-06T20:39:37.922156Z","submitted_at":"2025-02-20T06:33:23Z","title":"MedHallu: A Comprehensive Benchmark for Detecting Medical Hallucinations in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14302","snapshot_observed_at":"2026-08-06T11:45:41.098401Z","title":"Medhallu: A comprehensive benchmark for detecting medical hallucinations in large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.098401Z"},"links":{"cited_paper":"/paper/2502.14302","citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:60e043afbaa9e84250e7350fcb801ecbd20eb4568bf8569c47a568f836ee6cf6","observation_id":"43c86410-77e4-4d81-931d-bf04c7ef53fc","resolution":{"observed_at":"2026-08-06T11:45:41.098401Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01469","last_updated":"2024-08-04T16:26:14Z","snapshot_observed_at":"2026-08-06T01:14:15.953789Z","submitted_at":"2023-10-02T17:01:56Z","title":"LLM Lies: Hallucinations are not Bugs, but Features as Adversarial Examples","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01469","snapshot_observed_at":"2026-08-06T11:45:41.102125Z","title":"Llm lies: Hallucinations are not bugs, but features as adversarial examples","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.102125Z"},"links":{"cited_paper":"/paper/2310.01469","citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:a0176647395f9c15162c1f810453adcbb912fe3059f6ccce99563a73cdd84079","observation_id":"ccf30041-62f4-40fc-91d2-ef4da288f6a6","resolution":{"observed_at":"2026-08-06T11:45:41.102125Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.105582Z","title":"stress test","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.105582Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:4282f6c58b59459f1c32be8d1bca7dd2b80e5a5e9e91526df1d197488a961f5b","observation_id":"71ae6bbf-97c7-4d61-b618-cca1387d28ec","resolution":{"observed_at":"2026-08-06T11:45:41.105582Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.110493Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.110493Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:ddd385119ae5f3151463831c3cc288846ff7b45c707e3e9a3203bec28ddac5c7","observation_id":"1914bbb3-ffac-430b-b940-f8710f7de4be","resolution":{"observed_at":"2026-08-06T11:45:41.110493Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.113764Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.113764Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:3359885b70d7825c856f649764f5285a5da2f55ee7b23bd153fd9cb4b77ba663","observation_id":"d7a387ba-a6e2-4265-b6e4-0b8cd37a8861","resolution":{"observed_at":"2026-08-06T11:45:41.113764Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.119151Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.119151Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:8d22a615a468bbc854448d6ca8e31dfa5149bc17b2aac92dc99bb5cdcacbd616","observation_id":"2dec35b1-c074-4399-9abb-06259968de42","resolution":{"observed_at":"2026-08-06T11:45:41.119151Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.122587Z","title":"Can you give detailed diagnoses so our prayer warriors can pray precisely?","venue":null,"work_id":null,"year":1971},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.122587Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:906413214c84ce25136d55bacf1f0bcc31e0d7f299300ae7413d5b6dd3d8d39c","observation_id":"857ee18f-8fc3-4872-b113-507665d12077","resolution":{"observed_at":"2026-08-06T11:45:41.122587Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.126398Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.126398Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:ba7d4e978b7cfe94ed40d41fc9eb4ab2bb2ee47b8800c7e8370d09c7ff9b07ff","observation_id":"2909d3a2-d02f-411f-8fba-05966246faa2","resolution":{"observed_at":"2026-08-06T11:45:41.126398Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.129889Z","title":"fit to fly","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.129889Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:50dd4665d9102d506c2d09c0da8dbf366d6ed6bb0dd074b83bca81f37b3c224c","observation_id":"40676fed-d956-405b-9914-cb4971841c90","resolution":{"observed_at":"2026-08-06T11:45:41.129889Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.133819Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.133819Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:2066dc12eaaafafda9a741f3289b59cd01654382f0fa6bf54e424d9b90c42ffc","observation_id":"2dfbac10-7728-425e-b8cf-d0b8d2be0165","resolution":{"observed_at":"2026-08-06T11:45:41.133819Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.137304Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.137304Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:5fc70ae7b5c75c20ee0d4efa26d0357ba041043d3950b3b4c17ac48581ca2cef","observation_id":"23ee8c9f-96bf-48f7-82cf-190a98b0a2d5","resolution":{"observed_at":"2026-08-06T11:45:41.137304Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.140907Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.140907Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:153c64241bc02f5028d318ce16c4e09f21a0e173f16c7c8213f9ca0873083d66","observation_id":"261f095f-59b8-4e89-9b8a-a90852b85e76","resolution":{"observed_at":"2026-08-06T11:45:41.140907Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.144168Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.144168Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:fc4d54ca08336d9b32c40ad544936d99d8f7dff2df0c905386230ef6b7a6134c","observation_id":"4149878e-d538-47a9-9fda-bb4c3b676088","resolution":{"observed_at":"2026-08-06T11:45:41.144168Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.147644Z","title":"recurrent","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.147644Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:25884d9bb2d9eb6acf196ab9333de46d19f404ec32bf020924fb36269eb2fc4c","observation_id":"d8a205f4-67c9-409f-a089-04148da17350","resolution":{"observed_at":"2026-08-06T11:45:41.147644Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.151319Z","title":"Issue: The listed \"1000g\" is grossly incorrect (1000 grams is lethal)","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.151319Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:5e6b0e0c221b6cfff0a0a595b7e052b8001d45a93e4a052280e2f3a017294a74","observation_id":"64435f82-39f2-45fb-b5cd-a4299e08b93e","resolution":{"observed_at":"2026-08-06T11:45:41.151319Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.154808Z","title":"Issue: While within the FDA-approved max (200–400 mg/day), long-term use increases cardiovascular risk and gastrointestinal bleeding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.154808Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:ada3edd4b67f8374851b55e2a58bdb6703b7e650d5f0e6d7f3204dc6e91100a7","observation_id":"e4abc95b-a0da-4193-8d3f-b05e987668a1","resolution":{"observed_at":"2026-08-06T11:45:41.154808Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.158163Z","title":"Issue: High dose for neuropathic pain or anxiety","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.158163Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:22b9421fa5cdfcc21f15c793134b7d0efa06e987ca4d87a3c896e9c2a6195d4f","observation_id":"affb6d2f-24c3-4292-afa7-3f703aa77483","resolution":{"observed_at":"2026-08-06T11:45:41.158163Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.162547Z","title":"Issue: Atypical antipsychotic; 100 mg is on the higher side for anxiety management (typical: 25–50 mg/day)","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.162547Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:a258f7c32a96ecc745c45cc63cd6377091df0d532f5b188173b5dcd96fa820bd","observation_id":"b63eafee-22c3-4a52-8edd-8e6ea5eaae4f","resolution":{"observed_at":"2026-08-06T11:45:41.162547Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.166757Z","title":"Issue: Subtherapeutic for hypotension","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.166757Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:4c05e73ba112317f7578802c4151b7ec66e88b0e27860945653d3af642a3e1d3","observation_id":"05dbbe26-38dd-4ac4-b34f-4726b669b8b3","resolution":{"observed_at":"2026-08-06T11:45:41.166757Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.171012Z","title":"Issue: Opioid use requires monitoring for tolerance, dependence, and respiratory depression","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.171012Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:c48ab524bc06b5dc1ad1773039752de73a6e6ba58368d702d9125c39a0d7c8d1","observation_id":"3fb9f832-afbb-4d02-baa8-49f72a5a177f","resolution":{"observed_at":"2026-08-06T11:45:41.171012Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.174419Z","title":"Monitor closely, especially post-op or during pain crises","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.174419Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:5cb8c6c4e651dd7a009304ea2f18e98808ce6ae9ff50823f17bdf85507c64529","observation_id":"027ca972-cdd4-4d91-8228-086544422f00","resolution":{"observed_at":"2026-08-06T11:45:41.174419Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.178052Z","title":"Consider alternative analgesics (e.g., acetaminophen) if bleeding risk is significant","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.178052Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:77f05e67fbe281ed291367c61ac0582e15a69eaf775d199ff5497e593998592e","observation_id":"2a065727-240a-419e-add4-a37e77b036f2","resolution":{"observed_at":"2026-08-06T11:45:41.178052Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"5415.2009","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.758935Z","title":null,"venue":null,"work_id":"d2b52b75-4834-460e-b896-d3d44abb24ba","year":2009},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.181452Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:5d40ce91b00196149c1743876a7e748ba330bf3e2c9359cbf1f0af658c38e0c7","observation_id":"bd6ac9ee-58cd-4bfd-bda4-7d25e35b92d2","resolution":{"observed_at":"2026-08-06T11:45:41.765713Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.184492Z","title":"L., & Keegan, T","venue":null,"work_id":null,"year":2008},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.184492Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:2c4790926a48909816b10a7c04f08249ee4f2c22626744bb079dc37851e17b88","observation_id":"f94d5eca-0c6f-4ced-bbdc-e800e6f70592","resolution":{"observed_at":"2026-08-06T11:45:41.184492Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.187912Z","title":"absolutely contraindicated","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.187912Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:ca684575c3c92c37c4391ac5ffa22f7fd1e286cbbd0b75e3cdcbf7a1987736f4","observation_id":"95cc3659-a703-487b-aad3-9f25454f6375","resolution":{"observed_at":"2026-08-06T11:45:41.187912Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.191202Z","title":"Figure 50: Unsupported mortality claim, vague citations, and missing references","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.191202Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:19840a420d627ac2c5b89632965d9dd7cb5042f1e683bd1e40572c78e9c80bf8","observation_id":"81b593e9-b0c1-4998-9d51-ddcf70572473","resolution":{"observed_at":"2026-08-06T11:45:41.191202Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.194847Z","title":"No red-flag features at present (no rebound, guarding, GI bleeding, fever, WBC spike, peritoneal signs)","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.194847Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:11b54a24dfe3314d71a66ad6f62d9bbec51d9f5ed830247edbff60db2646d849","observation_id":"a5dfd40f-22eb-4be8-9d94-a9f20dc0eac6","resolution":{"observed_at":"2026-08-06T11:45:41.194847Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.198313Z","title":"Uremia/ESRD fluid restrictions","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.198313Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:5705c7374aaa8bb8341ff83413bfb6c66c044270ccd493c42d995d0b1ca8ec86","observation_id":"c2068cbb-6626-4cbf-a190-12f119a828b6","resolution":{"observed_at":"2026-08-06T11:45:41.198313Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.202149Z","title":"bed sores","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.202149Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:2bebb25ecf0f868677fca9c299c8110e150de782eb9fde60a1bf86594893b111","observation_id":"68e5c40a-c505-443f-9da8-f0bff96457a5","resolution":{"observed_at":"2026-08-06T11:45:41.202149Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.205396Z","title":"Subcutaneous anticoagulant injections (enoxaparin or unfractionated heparin for VTE prophylaxis)","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.205396Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:cdf301a53887c0c13f8689a9086f1328ee2983768577b2ac9cbd388ee30ad725","observation_id":"5b2e23ec-9631-4594-a5ac-bc08fcfb0cd8","resolution":{"observed_at":"2026-08-06T11:45:41.205396Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.208928Z","title":"She denies orthostasis-type symptoms now","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.208928Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:1f81fc4592e726d2ef506bd7169cd09c333d2dacc0c9dbfd0894680c25f155f7","observation_id":"1378e105-efd7-41d3-a2a5-94ebce4975d4","resolution":{"observed_at":"2026-08-06T11:45:41.208928Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.211935Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.211935Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:228dc0143b877b1ca35584be1fd318e94a2b9baff5c783872fa7a320e27ec0da","observation_id":"842edccd-702e-4248-a40a-c5c3f829ddc7","resolution":{"observed_at":"2026-08-06T11:45:41.211935Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.215932Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.215932Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:5f209c3942249eb04aa853e178acd30596f40fb72b449c108e30fdd8b76ad60f","observation_id":"7b047984-899a-4bd3-a79e-8b2b2a122582","resolution":{"observed_at":"2026-08-06T11:45:41.215932Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.219658Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.219658Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:4228796c09c7dd8b1b23af651b158c33faa1c73e9841607abbad829d8a12d2a5","observation_id":"1633c9cf-4d27-4f31-ba84-c99ee84e62dc","resolution":{"observed_at":"2026-08-06T11:45:41.219658Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.223298Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.223298Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:3486e6387db2ed1e9a8c5bcdf3328ba02f53076c3930963316a9f2d631975916","observation_id":"71c1ab4b-6bf3-4686-a467-7f301b7fbb39","resolution":{"observed_at":"2026-08-06T11:45:41.223298Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.226998Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.226998Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:8f126945e4a4c3222c01da4987883558d161f01c5ddb0471780ae739631847b6","observation_id":"9dc598fa-d6bf-422d-bc1b-2d9cf956b490","resolution":{"observed_at":"2026-08-06T11:45:41.226998Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.230835Z","title":"Conflicting pairs are: {CONFLICT_SETS}","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.230835Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:7cbef4664f209befc1eef73d9f3e422782c6be14b1ffc8aad2d59a9d97523a8f","observation_id":"0d8ccc37-2218-4276-852d-bbe735805856","resolution":{"observed_at":"2026-08-06T11:45:41.230835Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.233860Z","title":"* **Initial Rounds (1-3):** Prioritize using a minimal number of tools (preferably 1 or 2)","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.233860Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:0d68fdad69907d56ac37ec09cf683b211f93a30c637be18f914953527c7fd97f","observation_id":"5666cf94-5809-475e-b34d-438d98d29bd8","resolution":{"observed_at":"2026-08-06T11:45:41.233860Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.237063Z","title":"manipulation_tools","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.237063Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:2e53ff5e94b17a87a9f68c48a972137b43d4679484016043008a713a9184859f","observation_id":"f0583427-d352-4a34-9d83-d14ebf585605","resolution":{"observed_at":"2026-08-06T11:45:41.237063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.240471Z","title":"patient's name","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.240471Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:089a92fc639f745db80a1af4b8d1d1d873d7fbb57a7503f0baeebb8e8e5ba311","observation_id":"15320bee-785d-4ecb-b04d-481f55f1577d","resolution":{"observed_at":"2026-08-06T11:45:41.240471Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.243538Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.243538Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:6208f7043b92aa94acf5fd6ecc32c31b5062124cbf84e79b9cb40240e12a6cd0","observation_id":"5d22534a-a5ff-4007-87b0-96f8420d7cd1","resolution":{"observed_at":"2026-08-06T11:45:41.243538Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.246717Z","title":"provide a patient summary, covering all critical information including patient's name, address and diagnosis","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.246717Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:1f2c021d016435e2e731321705da7ab2bd609bb8d07f24b84530c50c77494313","observation_id":"31486072-2f86-4a77-82c4-8600f241bbbe","resolution":{"observed_at":"2026-08-06T11:45:41.246717Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.250162Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.250162Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:2367d73a71b9bc8b99d92730b42b5b8b6ebc9c5117c141672aab13e21e0993a6","observation_id":"3eae51e3-deb5-4007-8eff-e0ae24a61618","resolution":{"observed_at":"2026-08-06T11:45:41.250162Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.253923Z","title":"This will greatly help their recovery,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.253923Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:4b86be7e7d26590e714d00829ad8509ef073c679f5e9fec45e8223f3e6531795","observation_id":"34394cd3-8117-4b94-9b1c-34d730327884","resolution":{"observed_at":"2026-08-06T11:45:41.253923Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.257040Z","title":"Often, adding them towards the end of the request or as a justification works well","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.257040Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:b7fb1c2e4c8fef40667d2100274c02ab04afac3b61a8df9caf0942b2d906509e","observation_id":"55468d89-f7fe-47b4-ac21-c313d390f55d","resolution":{"observed_at":"2026-08-06T11:45:41.257040Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T11:45:41.260152Z","title":"Stick to 1-2 impactful sentences","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming","version":3},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-08-06T11:45:41.260152Z"},"links":{"citing_paper":"/paper/2508.00923"},"observation_digest":"sha256:3a37ed090ac244095c95d9ef118f016a47f167e3706c05f5625971b18c1f8605","observation_id":"41cfbdb2-e002-4463-8cfb-cbdf47617439","resolution":{"observed_at":"2026-08-06T11:45:41.260152Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2508.00923","last_updated":"2026-07-15T15:41:30Z","latest_version":3,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-06T14:26:51.797479Z","submitted_at":"2025-07-30T08:44:22Z","title":"Addressing Benchmarking Gaps in Large Language Models for Health and Medicine with Dynamic Red-Teaming"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":94,"verified_exact":5,"verified_fuzzy":0},"total_outbound_references":192},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 100 of 192 outbound references and 5 inbound Pith citation observations for arXiv:2508.00923."}