{"as_of":"2026-08-08T01:01:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:61663fd6e650b129bb078a0d5227eef5fbe24b15564e13dbfc8e8bcad51634b8","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":9,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":9,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":9,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":9,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T13:52:02.614223Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T09:47:59.494579Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2403.03744","last_updated":"2024-10-09T17:22:24Z","snapshot_observed_at":"2026-07-06T17:40:29.558340Z","submitted_at":"2024-03-06T14:34:07Z","title":"MedSafetyBench: Evaluating and Improving the Medical Safety of Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.03744","snapshot_observed_at":"2026-08-07T13:52:02.614223Z","title":"Medsafetybench: Evaluating and improving the medical safety of large language models.arXiv preprint arXiv:2403.03744, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20824","last_updated":"2025-05-27T07:34:40Z","snapshot_observed_at":"2026-08-07T23:40:07.104514Z","submitted_at":"2025-05-27T07:34:40Z","title":"MedSentry: Understanding and Mitigating Safety Risks in Medical LLM Multi-Agent Systems","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T13:52:02.614223Z"},"links":{"cited_paper":"/paper/2403.03744","citing_paper":"/paper/2505.20824"},"observation_digest":"sha256:f4a8035254ad85c25ab96d2017c998a12ff8f12836394b433003bcb3b66db5d8","observation_id":"7fbd9ea4-53be-4643-a757-3b4b035bec82","resolution":{"observed_at":"2026-08-07T13:52:02.614223Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.03744","last_updated":"2024-10-09T17:22:24Z","snapshot_observed_at":"2026-07-06T17:40:29.558340Z","submitted_at":"2024-03-06T14:34:07Z","title":"MedSafetyBench: Evaluating and Improving the Medical Safety of Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.03744","snapshot_observed_at":"2026-08-07T10:17:26.760644Z","title":"Medsafetybench: Evaluating and improving the medical safety of large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.11094","last_updated":"2025-10-30T06:22:33Z","snapshot_observed_at":"2026-08-07T10:11:06.747781Z","submitted_at":"2025-06-06T05:50:50Z","title":"The Scales of Justitia: A Comprehensive Survey on Safety Evaluation of LLMs","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T10:17:26.760644Z"},"links":{"cited_paper":"/paper/2403.03744","citing_paper":"/paper/2506.11094"},"observation_digest":"sha256:c3762af5dfe37c0109347dab89d625b7f621ebee55c893d9632b7eb17b14b298","observation_id":"01061d32-ffd0-4517-a929-7490bfdff1c3","resolution":{"observed_at":"2026-08-07T10:17:26.760644Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.03744","last_updated":"2024-10-09T17:22:24Z","snapshot_observed_at":"2026-07-06T17:40:29.558340Z","submitted_at":"2024-03-06T14:34:07Z","title":"MedSafetyBench: Evaluating and Improving the Medical Safety of Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.03744","snapshot_observed_at":"2026-08-06T21:05:18.696521Z","title":"Medsafetybench: Evaluating and improving the medical safety of large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02990","last_updated":"2025-07-01T18:00:04Z","snapshot_observed_at":"2026-08-06T20:57:35.466498Z","submitted_at":"2025-07-01T18:00:04Z","title":"`For Argument's Sake, Show Me How to Harm Myself!': Jailbreaking LLMs in Suicide and Self-Harm Contexts","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T21:05:18.696521Z"},"links":{"cited_paper":"/paper/2403.03744","citing_paper":"/paper/2507.02990"},"observation_digest":"sha256:86170e2a755e0d2191521d05ba3cad768d1671a48801de5c5ed2bf26a7ca36ae","observation_id":"ba9507fe-a727-4cc8-bb5b-495c55ce771a","resolution":{"observed_at":"2026-08-06T21:05:18.696521Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.03744","last_updated":"2024-10-09T17:22:24Z","snapshot_observed_at":"2026-07-06T17:40:29.558340Z","submitted_at":"2024-03-06T14:34:07Z","title":"MedSafetyBench: Evaluating and Improving the Medical Safety of Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.03744","snapshot_observed_at":"2026-08-06T14:13:06.342756Z","title":"Medsafetybench: Evaluating and improving the medical safety of large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19672","last_updated":"2025-07-25T20:52:58Z","snapshot_observed_at":"2026-08-07T12:02:14.124506Z","submitted_at":"2025-07-25T20:52:58Z","title":"Alignment and Safety in Large Language Models: Safety Mechanisms, Training Paradigms, and Emerging Challenges","version":1},"reference_index":111,"source":"arxiv_source","source_observed_at":"2026-08-06T14:13:06.342756Z"},"links":{"cited_paper":"/paper/2403.03744","citing_paper":"/paper/2507.19672"},"observation_digest":"sha256:f35748dbc0121b808b34001183a06b4674ccb060d0e87bd29482af57b32302a6","observation_id":"5b010107-49d5-488d-875b-9566461348f3","resolution":{"observed_at":"2026-08-06T14:13:06.342756Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.03744","last_updated":"2024-10-09T17:22:24Z","snapshot_observed_at":"2026-07-06T17:40:29.558340Z","submitted_at":"2024-03-06T14:34:07Z","title":"MedSafetyBench: Evaluating and Improving the Medical Safety of Large Language Models","version":5},"cited_work":{"arxiv_id":"2403.03744","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.03744","snapshot_observed_at":"2026-07-03T09:47:59.494579Z","title":"Haoan Jin, Jiacheng Shi, Hanhui Xu, Kenny Q","venue":null,"work_id":"bbf6840b-a020-4022-9a71-b4b4cc8138f7","year":2025},"citing_paper":{"arxiv_id":"2508.05132","last_updated":"2026-07-27T16:25:31Z","snapshot_observed_at":"2026-08-05T23:29:08.356955Z","submitted_at":"2025-08-07T08:10:14Z","title":"PrinciplismQA: A Philosophy-Grounded Approach to Assessing LLM-Human Clinical Medical Ethics Alignment","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-19T01:21:08.346848Z"},"links":{"cited_paper":"/paper/2403.03744","citing_paper":"/paper/2508.05132"},"observation_digest":"sha256:3bdc497be051ceffb1241a4d913bee7c778db6cb19386c0ee9c3c71f43f8a441","observation_id":"a5282501-aef6-411a-abc6-5cd8f01c1293","resolution":{"observed_at":"2026-05-19T01:21:57.196991Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.03744","last_updated":"2024-10-09T17:22:24Z","snapshot_observed_at":"2026-07-06T17:40:29.558340Z","submitted_at":"2024-03-06T14:34:07Z","title":"MedSafetyBench: Evaluating and Improving the Medical Safety of Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.03744","snapshot_observed_at":"2026-08-03T19:18:58.009540Z","title":"AMA augmented intelligence research: physician sentiments around the use of AI in health care: motivations, opportunities, risks, and use cases: shifts from 2023 to 2024","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2512.01241","last_updated":"2026-07-13T18:47:38Z","snapshot_observed_at":"2026-08-05T06:56:04.618729Z","submitted_at":"2025-12-01T03:33:16Z","title":"First, do NOHARM: a medical safety benchmark and randomized study of physician and AI teaming on clinical consultations","version":4},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-03T19:18:58.009540Z"},"links":{"cited_paper":"/paper/2403.03744","citing_paper":"/paper/2512.01241"},"observation_digest":"sha256:c331c8782e501a1a3ab6c953aba4b51f0270b985ea444cd33a43a6a0eb532abc","observation_id":"2ffd966b-8a2b-4a44-b892-4d7432998b18","resolution":{"observed_at":"2026-08-03T19:18:58.009540Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.03744","last_updated":"2024-10-09T17:22:24Z","snapshot_observed_at":"2026-07-06T17:40:29.558340Z","submitted_at":"2024-03-06T14:34:07Z","title":"MedSafetyBench: Evaluating and Improving the Medical Safety of Large Language Models","version":5},"cited_work":{"arxiv_id":"2403.03744","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.03744","snapshot_observed_at":"2026-07-03T09:47:59.494579Z","title":"Haoan Jin, Jiacheng Shi, Hanhui Xu, Kenny Q","venue":null,"work_id":"bbf6840b-a020-4022-9a71-b4b4cc8138f7","year":2025},"citing_paper":{"arxiv_id":"2512.20983","last_updated":"2026-04-07T04:41:41Z","snapshot_observed_at":"2026-08-04T14:29:40.044760Z","submitted_at":"2025-12-24T06:17:21Z","title":"Automatic Replication of LLM Mistakes in Medical Conversations","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-16T20:25:24.722562Z"},"links":{"cited_paper":"/paper/2403.03744","citing_paper":"/paper/2512.20983"},"observation_digest":"sha256:224d7ac9fbe4c90ebb2ecb7a0609f9686373180b7b69a26ee9a384ebe982356c","observation_id":"62aaddaa-7d92-44e2-b565-92f9e4826dcd","resolution":{"observed_at":"2026-05-16T20:28:24.180954Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.03744","last_updated":"2024-10-09T17:22:24Z","snapshot_observed_at":"2026-07-06T17:40:29.558340Z","submitted_at":"2024-03-06T14:34:07Z","title":"MedSafetyBench: Evaluating and Improving the Medical Safety of Large Language Models","version":5},"cited_work":{"arxiv_id":"2403.03744","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.03744","snapshot_observed_at":"2026-07-03T09:47:59.494579Z","title":"Haoan Jin, Jiacheng Shi, Hanhui Xu, Kenny Q","venue":null,"work_id":"bbf6840b-a020-4022-9a71-b4b4cc8138f7","year":2025},"citing_paper":{"arxiv_id":"2606.11740","last_updated":"2026-06-10T07:16:27Z","snapshot_observed_at":"2026-08-02T02:17:16.809620Z","submitted_at":"2026-06-10T07:16:27Z","title":"UniReason-Med: A Shared Grounded Reasoning Interface for 2D-to-3D Transfer in Medical VQA","version":1},"reference_index":172,"source":"arxiv_source","source_observed_at":"2026-06-27T10:21:12.782864Z"},"links":{"cited_paper":"/paper/2403.03744","citing_paper":"/paper/2606.11740"},"observation_digest":"sha256:139d6cd44233081056fe406eb1dc3ed11c8ccd8c56230943ca4a941e651bdc79","observation_id":"b8b3ca18-48ee-4c58-9f06-4e76d8b34dbe","resolution":{"observed_at":"2026-07-03T09:47:59.495945Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.03744","last_updated":"2024-10-09T17:22:24Z","snapshot_observed_at":"2026-07-06T17:40:29.558340Z","submitted_at":"2024-03-06T14:34:07Z","title":"MedSafetyBench: Evaluating and Improving the Medical Safety of Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.03744","snapshot_observed_at":"2026-07-11T11:38:20.987334Z","title":"Towards Safe and Reliable Large Language Models for Medicine,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.04907","last_updated":"2026-07-06T10:36:04Z","snapshot_observed_at":"2026-08-03T16:25:09.484049Z","submitted_at":"2026-07-06T10:36:04Z","title":"Medi-Gemma: A Hybrid Clinical Decision Support System Integrating Deterministic EMR Analytics and Retrieval-Augmented Generation","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-07-11T11:38:20.987334Z"},"links":{"cited_paper":"/paper/2403.03744","citing_paper":"/paper/2607.04907"},"observation_digest":"sha256:84bf1f59d5b98eb3e55de3bb9b225f669e659a493855751539a053015a38f1e3","observation_id":"5200e208-8256-4355-8e45-00976e832edd","resolution":{"observed_at":"2026-07-11T11:38:20.987334Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2403.03744/citation-record","integrity":"/paper/2403.03744/integrity","json":"/paper/2403.03744/citation-record.json","paper":"/paper/2403.03744"},"outbound":[],"paper":{"arxiv_id":"2403.03744","last_updated":"2024-10-09T17:22:24Z","latest_version":5,"primary_category":"cs.AI","snapshot_observed_at":"2026-07-06T17:40:29.558340Z","submitted_at":"2024-03-06T14:34:07Z","title":"MedSafetyBench: Evaluating and Improving the Medical Safety of Large Language Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 9 inbound Pith citation observations for arXiv:2403.03744."}