{"as_of":"2026-08-07T09:21:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:55ab9fcb963afdba09e4e48a18869d34425c9eab9e164ea5c80bfd8e41010cef","coverage":[{"denominator":61,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":61,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T14:42:18.697628Z","state":"measured"},{"denominator":62,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":62,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-29T23:18:59.283834Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-06-29T23:24:01.661042Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"cited_work":{"arxiv_id":"2507.18143","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.18143","snapshot_observed_at":"2026-06-29T23:24:01.661042Z","title":null,"venue":null,"work_id":"982cd8a3-f618-4737-99bb-a3eadb99e508","year":2025},"citing_paper":{"arxiv_id":"2605.25273","last_updated":"2026-05-24T21:59:32Z","snapshot_observed_at":"2026-07-06T23:35:16.647496Z","submitted_at":"2026-05-24T21:59:32Z","title":"LLM-as-a-Judge in Healthcare: A Scoping Analysis of Applications, Methods, and Human Alignment","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-29T23:18:59.283834Z"},"links":{"cited_paper":"/paper/2507.18143","citing_paper":"/paper/2605.25273"},"observation_digest":"sha256:6de488d6f249c7b4eb8bcc100d373fe4b3c8d13f914061c26aea63f5b684abee","observation_id":"19b33552-b332-4037-a819-f9cd47f40acb","resolution":{"observed_at":"2026-06-29T23:24:01.662324Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2507.18143/citation-record","integrity":"/paper/2507.18143/integrity","json":"/paper/2507.18143/citation-record.json","paper":"/paper/2507.18143"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:20.339491Z","title":"& Topol, E","venue":null,"work_id":"c71ab94c-6e6a-4c83-bbc5-69ac8ac4238f","year":2025},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.327944Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:0addb424e1110cdad48a45252503b927724aed81b78c0af4559f78f48ac5b0d7","observation_id":"d00093d7-01e1-48eb-bb8c-7a925d5f04f4","resolution":{"observed_at":"2026-08-06T14:42:20.345475Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:20.314764Z","title":null,"venue":null,"work_id":"a109c08f-6db5-4c4a-9d87-83497538f896","year":2024},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.333246Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:e4267cff2098e1d0603f76b55bd6b986d952ba7d818d2512e9b1fd99f7aeaf6d","observation_id":"cbade836-7e62-48fc-a0c6-56b58dc1a4dc","resolution":{"observed_at":"2026-08-06T14:42:20.324802Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:20.295598Z","title":"& Taylor, R","venue":null,"work_id":"b10c1c01-4467-4a84-827a-3f7d1fd6ecef","year":2024},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.338407Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:104e050d1f4ce3f948a5eced40670bd5ba6e7820a6e105e47ff2b080edac292b","observation_id":"8cb3275f-2cd7-4c1e-87aa-50b5aa0d57b9","resolution":{"observed_at":"2026-08-06T14:42:20.301557Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:20.273254Z","title":null,"venue":null,"work_id":"4b22c082-572a-4e61-b2d1-5f2e08cabf3d","year":2023},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.348108Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:57445597c5899d740917f98ada620ba5331beb93a9a6679b1b490a6758b2de99","observation_id":"a07a9f0c-7f11-4a62-aac3-8816d3a708b3","resolution":{"observed_at":"2026-08-06T14:42:20.279241Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:20.253904Z","title":null,"venue":null,"work_id":"344f0fbb-7082-4803-af82-3e2bdcb1ed79","year":2024},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.353270Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:d55b08665402d3dd4ba9fd555d243d974100bf04d9503125dc4e420875102b9d","observation_id":"7ab591a6-d8c3-434a-a2e7-c59205dfc5b0","resolution":{"observed_at":"2026-08-06T14:42:20.258544Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:20.230721Z","title":null,"venue":null,"work_id":"76bc76c8-1bab-4b11-b1b5-26b743f3cf18","year":2023},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.358167Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:7ae83c4257b2e077df7753c428cd3325451c12cec3dfef26d05a23c2a51de669","observation_id":"d1145283-b369-458e-8b44-981598025c41","resolution":{"observed_at":"2026-08-06T14:42:20.238457Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:20.211142Z","title":"Large language models improve clinical decision making of medical students through patient simulation and structured feedback: a randomized controlled trial","venue":null,"work_id":"551b1cf4-863c-4cc2-aa5b-dbbdc3c6d2c0","year":2024},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.364769Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:599b5268b1283c0c7bbcefec4c3e0e856956013a4447c422638750aff50023b2","observation_id":"3bfe12e4-5e2b-4491-a1f7-701e880d3e76","resolution":{"observed_at":"2026-08-06T14:42:20.216975Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16452","last_updated":"2023-11-28T03:16:12Z","snapshot_observed_at":"2026-08-07T08:26:36.845011Z","submitted_at":"2023-11-28T03:16:12Z","title":"Can Generalist Foundation Models Outcompete Special-Purpose Tuning? Case Study in Medicine","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16452","snapshot_observed_at":"2026-08-06T14:42:18.370147Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.370147Z"},"links":{"cited_paper":"/paper/2311.16452","citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:8c7a1c336e2245fdad36767559206965b1a59efd4ebc99c4d547602c68b70796","observation_id":"9850c560-7d20-4d05-8d5e-bf9bf6e0f933","resolution":{"observed_at":"2026-08-06T14:42:18.370147Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:20.184980Z","title":"& Petro, J","venue":null,"work_id":"35a267a1-35bf-4ce4-93de-ec26c3da4a80","year":2023},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.376914Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:ea8324e61cd2476113a40e2758dc6edef4533ebc68d793c0d91d2cee54222fcf","observation_id":"b520f252-391d-40a0-bada-e08f7a2c7906","resolution":{"observed_at":"2026-08-06T14:42:20.195521Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:20.162757Z","title":null,"venue":null,"work_id":"e870887f-4cec-4113-acaf-ce212eb0564c","year":2024},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.382624Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:e3ed8751c536d2ed61142c9bb70fc0eef119c4c82213aeaa0bb0f8c3666641f4","observation_id":"a43450ec-d55d-429d-a722-69591d5051e0","resolution":{"observed_at":"2026-08-06T14:42:20.169555Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:20.138068Z","title":null,"venue":null,"work_id":"91280004-53b7-49b5-b5fd-29eb0d7c8d24","year":2023},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.388422Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:eebc93842c50aab29f6ee8775bba2e8e9687e5709013db6c4fedbbf271b7c015","observation_id":"b9cd8862-9e6b-4482-940d-86bda2612c4e","resolution":{"observed_at":"2026-08-06T14:42:20.147090Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:20.105559Z","title":null,"venue":null,"work_id":"f1944e60-8789-4dda-a4b0-c9e4f5e4be21","year":2017},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.396040Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:c5c8e5bfaf9185ee79a8ab78136ab7eb6f8872643a23207a26bee795ced37588","observation_id":"6d58057c-469f-4d10-a7ea-6dfd1cfea59f","resolution":{"observed_at":"2026-08-06T14:42:20.112418Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:20.072390Z","title":"& Chow, C","venue":null,"work_id":"f72a4ae9-a2c4-405f-a01b-60c4e5a5063d","year":2020},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.401352Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:3121f60d7b6219c3de6448dfe6ec450ffd02ae1bbda7b82f278b5d20489fddd0","observation_id":"a0268c01-e7f3-4dd3-b561-f2d9bc5a82d7","resolution":{"observed_at":"2026-08-06T14:42:20.078979Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:20.043706Z","title":"& Dussault, G","venue":null,"work_id":"2e5f75ca-519e-4e2c-a41a-422de8a775ab","year":2019},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.407255Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:a5a939159cca2e665c20830efab821e428a7e3ad30204a832958e2e35d93d536","observation_id":"4258786c-ff17-458e-9d9a-3d0308c087c1","resolution":{"observed_at":"2026-08-06T14:42:20.049365Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:20.018284Z","title":null,"venue":null,"work_id":"01a91bb1-5f9f-4016-926e-ff8b8ee91aa9","year":2025},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.414911Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:2bdb1d8493477fc4b51c7201d4f50c5f93732fe212fae9eea637cc75f33fcbed","observation_id":"df319221-9d75-4762-b780-023922aa6af9","resolution":{"observed_at":"2026-08-06T14:42:20.027482Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:19.989177Z","title":"S., Link, K","venue":null,"work_id":"71917aa7-2db7-4ffe-a639-410529e68c52","year":2024},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.423293Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:4174df2c5441891963438c02b2fde65d35bc5a305de69c56a88cf51ce7de8438","observation_id":"4144b980-0807-4326-83eb-902099f163dd","resolution":{"observed_at":"2026-08-06T14:42:19.996200Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:19.966467Z","title":"A., Lester, J","venue":null,"work_id":"84593340-d652-4915-bf10-09bf63b13fd2","year":2023},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.428884Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:d32b2768e60f38ecf573926b558eea3258aacda7d8eba75d796d79cc295aabe1","observation_id":"8b1b808f-b918-46e3-90b2-adf5a6666d7f","resolution":{"observed_at":"2026-08-06T14:42:19.975130Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:19.944304Z","title":null,"venue":null,"work_id":"3a2fdbc5-e0f0-40cc-92eb-53c097dde105","year":2024},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.433357Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:a787442e728c3318a8cd79e1a2d91f9f83cb4d49cfa85f7e8ff671f3c01ae81f","observation_id":"e9b44823-ec67-431b-95af-2be9221ba3cb","resolution":{"observed_at":"2026-08-06T14:42:19.951455Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.07314","last_updated":"2026-07-28T17:58:21Z","snapshot_observed_at":"2026-08-05T16:41:56.624766Z","submitted_at":"2024-09-11T14:44:51Z","title":"MEDIC: Comprehensive Evaluation of Leading Indicators for LLM Safety and Utility in Clinical Applications","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.07314","snapshot_observed_at":"2026-08-06T14:42:18.438834Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.438834Z"},"links":{"cited_paper":"/paper/2409.07314","citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:0bbdd200701517a9f7d873717004c248c8ab7e501fefc05975ddd426d2499bbe","observation_id":"194980ef-8d66-47ce-885c-05672c89046a","resolution":{"observed_at":"2026-08-06T14:42:18.438834Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:19.919483Z","title":null,"venue":null,"work_id":"157e492b-3623-4012-8965-f6b4797e1ec0","year":2023},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.444401Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:8440d72886a2e35ded5c35051c7999c8be15376e2d6bf9d329925f691757bc0d","observation_id":"283e7acc-25df-4505-9085-9bf869211366","resolution":{"observed_at":"2026-08-06T14:42:19.927167Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.07960","last_updated":"2025-05-25T02:19:37Z","snapshot_observed_at":"2026-07-30T08:34:47.046912Z","submitted_at":"2024-05-13T17:38:53Z","title":"AgentClinic: a multimodal agent benchmark to evaluate AI in simulated clinical environments","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.07960","snapshot_observed_at":"2026-08-06T14:42:18.449467Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.449467Z"},"links":{"cited_paper":"/paper/2405.07960","citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:ca1881798c63b751592c212a52daf6062f5a92a52257cc6b7ab5cd212644b236","observation_id":"b09057d3-9bd3-4014-bb7d-26f8b9c2eb2f","resolution":{"observed_at":"2026-08-06T14:42:18.449467Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:19.895458Z","title":null,"venue":null,"work_id":"fb2de950-78cd-4a8d-9e04-6e08f539edbe","year":2025},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.457258Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:cdbc7a5d1d6dfb3515235964001ee65c0922bf8f6d3eb824db19d6df0be51882","observation_id":"bbbbf14a-1c89-4f74-8a0b-584165d72658","resolution":{"observed_at":"2026-08-06T14:42:19.900924Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.21203/rs.3.rs-4139743/v1","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:18.745433Z","title":null,"venue":null,"work_id":"8827bba9-e135-45a9-af9a-ef5b1e7fb93f","year":2024},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.466737Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:1917e72d4aa2989568c31c646b56115e048efae6b495ebcf418f2e5dd6768890","observation_id":"54eaaba1-7eaa-4eb1-97b4-c421d08e0230","resolution":{"observed_at":"2026-08-06T14:42:18.753396Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:19.871941Z","title":null,"venue":null,"work_id":"42b6a868-6216-4746-879e-f9007204f6d4","year":2023},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.472116Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:3f31a73e9e3ce40164d662cea309ebad18d3804bf137be80780528e22eae7586","observation_id":"7100ea01-68b9-4964-9fa2-cf8de4a11d86","resolution":{"observed_at":"2026-08-06T14:42:19.881456Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:18.478987Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.478987Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:6674091b1cd2c1ac429652c5ea6fcc6a91ba15af4f8b9d1b55666f6896946d00","observation_id":"905063a4-eb7c-40f9-a37b-15af8cd2b7b2","resolution":{"observed_at":"2026-08-06T14:42:18.478987Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:19.832679Z","title":null,"venue":null,"work_id":"0d21bf03-d940-41d2-9ab1-947b5c8c819f","year":2024},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.483722Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:357c68036422054bce838ce828b126412fa86814cd9d6548c64d0382d9e46c21","observation_id":"30d1fa25-0606-4c56-873c-e0a04c54c838","resolution":{"observed_at":"2026-08-06T14:42:19.839199Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.13833","last_updated":"2024-08-25T13:36:22Z","snapshot_observed_at":"2026-07-06T19:05:40.623256Z","submitted_at":"2024-08-25T13:36:22Z","title":"Biomedical Large Languages Models Seem not to be Superior to Generalist Models on Unseen Medical Data","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.13833","snapshot_observed_at":"2026-08-06T14:42:18.489506Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.489506Z"},"links":{"cited_paper":"/paper/2408.13833","citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:0e973e540d1766cc098ac434fa542c1dc1d656c735b3ba0ff217d30a39859f71","observation_id":"f52043c0-b81a-4f05-9069-e414a76359e9","resolution":{"observed_at":"2026-08-06T14:42:18.489506Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:19.809515Z","title":null,"venue":null,"work_id":"94b950c4-4aa8-47f8-b174-d2bf333a115a","year":2023},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.497292Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:00871018621878a14bf98370c7268ac1df11de00d917be5d68c32ec350ba3e31","observation_id":"32ad7521-810e-4550-8e2f-19abbdb99079","resolution":{"observed_at":"2026-08-06T14:42:19.817052Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:19.784802Z","title":null,"venue":null,"work_id":"d84eb325-5d5d-4a67-9a0b-d6f1fe6f5039","year":2023},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.502209Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:c8b5cd7011f23a810439a0447e027d176387ab80073ead4af0e2dbbec83e58fb","observation_id":"127a8375-058a-45e2-ba9d-17c2f4ffb014","resolution":{"observed_at":"2026-08-06T14:42:19.790721Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:19.759031Z","title":null,"venue":null,"work_id":"0621ab5d-d8a8-4983-abb5-5343dc3c7aa8","year":2023},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.507177Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:440532556ce5306ea3fc0caae265f9ade688a8fabb459e225ae7e6d7032e2286","observation_id":"2c305b8a-1052-421a-8afb-cbb5ccd6ec79","resolution":{"observed_at":"2026-08-06T14:42:19.769605Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:19.725089Z","title":"A., Lingohr-Smith, M., Rogers, R., Lin, J","venue":null,"work_id":"95423842-46e3-43d0-a110-a4dc5ce280ef","year":2021},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.515205Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:de54469d12b962dc501bcb9f18cf8c874e096bd9279b55028f2903beef69c75b","observation_id":"af8f18bc-6c5c-4b95-84b3-8c981eaebaba","resolution":{"observed_at":"2026-08-06T14:42:19.731208Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:19.706132Z","title":null,"venue":null,"work_id":"73d1d080-b9e8-47e1-b06f-012bdfb55ecc","year":2025},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.519604Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:2ad2b4f1dc3f6d6103b9377cebec8e8b09ee3ea36348679eee43c19451736a63","observation_id":"6d6b0bb9-fb89-4bcf-85f7-e95a6da22a90","resolution":{"observed_at":"2026-08-06T14:42:19.712854Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-06T14:42:18.525401Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.525401Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:75e3b655596349788b4ae1228d39077f5839625e66b65a35369cf65bfc0600bf","observation_id":"9b5f2c17-8b58-47fa-9b71-d29d63ef46c0","resolution":{"observed_at":"2026-08-06T14:42:18.525401Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:19.687357Z","title":null,"venue":null,"work_id":"db8f4f71-7804-421f-8e4e-7a45bdac462a","year":2024},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.531488Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:270a44558b4b846bea847c30028b017c9bcc6c59f18defc51f7fc71c7bcc19d7","observation_id":"1a0ddae2-c61f-4989-99e2-13ed60bfe4f6","resolution":{"observed_at":"2026-08-06T14:42:19.693675Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.06142","last_updated":"2024-08-12T13:37:31Z","snapshot_observed_at":"2026-08-04T21:49:27.462796Z","submitted_at":"2024-08-12T13:37:31Z","title":"Med42-v2: A Suite of Clinical LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.06142","snapshot_observed_at":"2026-08-06T14:42:18.536520Z","title":"K., Raha, T., Khan, S","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.536520Z"},"links":{"cited_paper":"/paper/2408.06142","citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:116bec0e73db8e50e36c05897c2410ed116a6792feadfbaaba79ac97981736e4","observation_id":"e67a81bc-4767-41e5-ae98-d45116476693","resolution":{"observed_at":"2026-08-06T14:42:18.536520Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:18.542389Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.542389Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:0991b1823c105b7d5e54fdc3248c92d10088c70809749ea65bd306c425c7cea3","observation_id":"b116b51f-60ef-42bf-a9f9-b39101152c97","resolution":{"observed_at":"2026-08-06T14:42:18.542389Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:19.655350Z","title":null,"venue":null,"work_id":"a8bc32f1-6f75-485d-938a-1b4b4fefc379","year":2024},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.547648Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:e40a19ad2756055399842cd5717e36cac64e3850990d5c5728aa033e2c79dc04","observation_id":"91abcdb6-f3b3-422d-8517-ae17d85f68f3","resolution":{"observed_at":"2026-08-06T14:42:19.663539Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:18.553615Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.553615Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:5242bbd9f482f193b515d49a157f9a1d1b04f046a333ffba6bbf7058afaa0d4c","observation_id":"85188c7c-f502-4aad-828d-c21a2ebc916f","resolution":{"observed_at":"2026-08-06T14:42:18.553615Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:19.626026Z","title":null,"venue":null,"work_id":"2329c769-fb59-4992-965f-bc66d06cf3eb","year":2024},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.559307Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:19b46a38a2bd1804d0fdb54d63cbdfd2f061f3e6afa40aa1a4e7d891ad700fe7","observation_id":"238f3bf3-6cd0-4784-8f72-113d9b51bffa","resolution":{"observed_at":"2026-08-06T14:42:19.631293Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.09834","last_updated":"2024-11-19T21:04:38Z","snapshot_observed_at":"2026-08-06T23:57:52.404063Z","submitted_at":"2024-11-14T22:54:38Z","title":"A Benchmark for Long-Form Medical Question Answering","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.09834","snapshot_observed_at":"2026-08-06T14:42:18.565269Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.565269Z"},"links":{"cited_paper":"/paper/2411.09834","citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:e57ddd5490375e388a316baa5c92b7bc126ca4e065e889324071b968aa1e4523","observation_id":"733c0ab9-b198-48e8-a7fd-b0d6895220af","resolution":{"observed_at":"2026-08-06T14:42:18.565269Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:19.594925Z","title":null,"venue":null,"work_id":"e861fb1c-50bf-4875-9960-093978d06290","year":2025},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.571529Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:24ece9bd2ac58e183675d362322e0d86acfe29669bec451687cc145d16f451c4","observation_id":"73429537-e222-466a-a55f-98764968841c","resolution":{"observed_at":"2026-08-06T14:42:19.601340Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.04166","last_updated":"2023-02-13T16:19:05Z","snapshot_observed_at":"2026-08-05T08:58:00.258106Z","submitted_at":"2023-02-08T16:17:29Z","title":"GPTScore: Evaluate as You Desire","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.04166","snapshot_observed_at":"2026-08-06T14:42:18.577140Z","title":"& Liu, P","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.577140Z"},"links":{"cited_paper":"/paper/2302.04166","citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:405f142ffc4cdab27b2b3cb1f93fd90d9d72148f94b7844d074c409a6bfab33b","observation_id":"b656ce3a-8f40-43c1-8439-db8ba960243d","resolution":{"observed_at":"2026-08-06T14:42:18.577140Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.17887","last_updated":"2024-06-28T13:23:31Z","snapshot_observed_at":"2026-07-06T17:36:27.931373Z","submitted_at":"2024-02-27T21:01:41Z","title":"JMLR: Joint Medical LLM and Retrieval Training for Enhancing Reasoning and Professional Question Answering Capability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.17887","snapshot_observed_at":"2026-08-06T14:42:18.584585Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.584585Z"},"links":{"cited_paper":"/paper/2402.17887","citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:9d2bbb69f67b387cecbee9442f753c0bf4a3d9ef91444193398f2baf2abbe031","observation_id":"7a870550-04c4-4e4a-a3c1-716001ae2c5e","resolution":{"observed_at":"2026-08-06T14:42:18.584585Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:18.590156Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.590156Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:607c751b7684bbca4e7b7c11462d0176bf1d2e14bbf3543d806ecfd15cdd11f7","observation_id":"850d7db2-28cd-4daa-97ec-fc87bfbdcac5","resolution":{"observed_at":"2026-08-06T14:42:18.590156Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:19.561791Z","title":"Rouge: A package for automatic evaluation of summaries","venue":null,"work_id":"57021c8c-c6aa-40c5-bee3-d0499342f8d5","year":2004},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.596357Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:d38a3ba33e04546dc011a8912eb1178781272f4f88d2a3f77473b8323c55c6e2","observation_id":"bb43128b-c32b-4dc2-85f7-612df07471ef","resolution":{"observed_at":"2026-08-06T14:42:19.566562Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:19.541298Z","title":"& Zhu, W.-J","venue":null,"work_id":"fd8ccdcc-fe89-49a7-8cb7-80506c04556d","year":2002},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.602428Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:33f5fb3dad43d3b15704244d2f65bea74cbf367e60e89c99a78b97a799a69a96","observation_id":"0cb365e8-567e-4e8f-b125-853006d854d7","resolution":{"observed_at":"2026-08-06T14:42:19.546357Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:19.520299Z","title":"The unified medical language system (umls): integrating biomedical terminology","venue":null,"work_id":"cc12afb1-2cc9-4ae1-855b-29ff36e5ba0a","year":2004},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.608825Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:586f55b4705acb8e5dbd79967e8383963e2e3ebc7b0b959695f84840d089376c","observation_id":"2e5c27f2-ab9e-46c9-9892-4b01b10ffeb6","resolution":{"observed_at":"2026-08-06T14:42:19.528066Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1902.07669","last_updated":"2019-10-09T23:07:18Z","snapshot_observed_at":"2026-08-06T15:31:44.011204Z","submitted_at":"2019-02-20T17:28:51Z","title":"ScispaCy: Fast and Robust Models for Biomedical Natural Language Processing","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1902.07669","snapshot_observed_at":"2026-08-06T14:42:18.614111Z","title":"& Ammar, W","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.614111Z"},"links":{"cited_paper":"/paper/1902.07669","citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:8246943be59b05f43fcaf13a32ecc77172105a568a0564e6ed9399b830c3afd3","observation_id":"6133ece7-f6fe-4a77-9e9b-4196eab01b6e","resolution":{"observed_at":"2026-08-06T14:42:18.614111Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:19.503597Z","title":"& Duclos, C","venue":null,"work_id":"580cf3de-80d9-43c5-9f41-3d3f9838f9f7","year":2015},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.619442Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:cbcb59924a1717112dfabd219518ca568e71cb27ea1f974dc7d42311e26628b6","observation_id":"de0629ec-f942-4841-afa1-579c39959101","resolution":{"observed_at":"2026-08-06T14:42:19.509088Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:19.487679Z","title":"Owlready: Ontology-oriented programming in python with automatic classification and high level constructs for biomedical ontologies","venue":null,"work_id":"4de7d65e-e942-49c9-999f-5307aa457559","year":2017},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.624809Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:22ad56b8a3050571f012eed571ffc12fb301e79642505d9df907256dcdd8d2e5","observation_id":"88e426b7-3f18-457c-a08b-de8d3d112b1b","resolution":{"observed_at":"2026-08-06T14:42:19.492646Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:19.468284Z","title":"Unified Medical Language System (UMLS): 2024AB Full Release Files (2024)","venue":null,"work_id":"ed7fe253-3cb6-4ecc-a2b5-42869086c167","year":2024},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.630257Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:62e0dddb17d4bd36f5cef4e9fdec604b912c084660eea2756f26e334e305b9e1","observation_id":"b556e891-6910-45b7-9227-35b94625b3d3","resolution":{"observed_at":"2026-08-06T14:42:19.474078Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:19.452118Z","title":"Nltk: the natural language toolkit","venue":null,"work_id":"d0dde8cf-0971-41bd-8823-74376e5beb3c","year":2006},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.637022Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:cb988f26a2c748fc7c26d7783c69b1f2d345228fc4b9ee2b04d6d4756bce9b2d","observation_id":"39c37b99-6a8e-48bf-8e40-05f3efd272b5","resolution":{"observed_at":"2026-08-06T14:42:19.457002Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:19.427954Z","title":null,"venue":null,"work_id":"0e73ddf3-e4d1-4a1d-ba79-b5dc9fa682e3","year":1995},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.645060Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:a2fb1e86ad4040a34d9f06587d24ad202ac9aeba082b7962fbbe251f8206b751","observation_id":"d7074ab6-9d76-42e6-ad9c-628e233c462f","resolution":{"observed_at":"2026-08-06T14:42:19.433360Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:19.403532Z","title":null,"venue":null,"work_id":"e43ab575-dd41-4c41-800c-82183dd97252","year":2021},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.652634Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:1382065a80a74e5262e08b58b328e6b7d386575a45b0e52f28f7dab2e3d8a0a9","observation_id":"611bcb64-c907-412e-b97e-675a73de8fda","resolution":{"observed_at":"2026-08-06T14:42:19.412058Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:19.379655Z","title":null,"venue":null,"work_id":"3fa7f51f-bb9e-4c40-b5d1-0f861265c6fb","year":2003},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.663408Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:c9f1604e7f674db4b85f3c1f0e9e0d05696e1af6d3b4db789a515dca31a7ecc6","observation_id":"21637a13-bd05-44d4-b07f-63821aae5d57","resolution":{"observed_at":"2026-08-06T14:42:19.385838Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:19.349823Z","title":null,"venue":null,"work_id":"1c0e3c55-b67e-4be0-9aff-6309220a2dc7","year":2011},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.670150Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:32747da79956a58b21ff6b619b5b67a8b8d2fac5e7517914ea17925e03262fb9","observation_id":"21272b7e-acc2-410c-8499-524d04da7d5d","resolution":{"observed_at":"2026-08-06T14:42:19.365142Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:19.326570Z","title":"- 1-2: The student’s answer shows partial understanding but contains notable misinterpretations","venue":null,"work_id":"41cb0481-510a-4c55-81ed-8a4a753eb711","year":null},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.676520Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:f80536ba0c87ed8808228a8af1e45412be5d467121f50244e67775aead59b59f","observation_id":"25b0ab7b-26fb-4b0f-acf9-01c3e1c304f1","resolution":{"observed_at":"2026-08-06T14:42:19.332237Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:19.309622Z","title":"- Score low if the reasoning lacks clarity or is inconsistent with medical principles","venue":null,"work_id":"b2c43789-da2c-4736-b5b6-62e043e41462","year":null},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.681719Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:9fb3646667228f935956c5c77ee235a33ee4b46dd9cd2c933edde982c8d08c4c","observation_id":"3cb5bd30-414a-4147-af87-b03d56049fd2","resolution":{"observed_at":"2026-08-06T14:42:19.315183Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:19.291080Z","title":"- A lower score should reflect the severity and frequency of factual errors","venue":null,"work_id":"7f602ad1-a99a-4e60-894d-76cc355d1a1c","year":null},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.686903Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:65e2947d6a3609580c8ff1ca23b21d0cc44d901027d52e522e0b282af46503ed","observation_id":"cceefff4-bf91-4f9d-9ca7-aebe4c019f74","resolution":{"observed_at":"2026-08-06T14:42:19.296570Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:19.270910Z","title":"- A perfect score requires complete neutrality and sensitivity","venue":null,"work_id":"52da1c24-a3bc-4959-b7fc-633870974081","year":null},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.692350Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:ee1d6e27507a6383d5b126b6e82974becdb06e9978b47cf95bf2694d209868c6","observation_id":"0f45edbf-6d13-44e2-907e-40830b13a740","resolution":{"observed_at":"2026-08-06T14:42:19.277039Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:42:19.249669Z","title":"- Perfect scores require clear evidence of safety-oriented thinking","venue":null,"work_id":"3228d153-bcaa-4378-868b-15d57b7ca003","year":null},"citing_paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support","version":2},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:18.697628Z"},"links":{"citing_paper":"/paper/2507.18143"},"observation_digest":"sha256:dcdbac9406422b8fbb9141ec1fd57c4d8aa50821575eea02083839fe3552e367","observation_id":"7b71bcc7-b87a-49af-b57e-7544a5405a40","resolution":{"observed_at":"2026-08-06T14:42:19.256427Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.18143","last_updated":"2025-07-25T06:40:44Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-06T16:38:03.240112Z","submitted_at":"2025-07-24T07:06:30Z","title":"HIVMedQA: Benchmarking large language models for HIV medical decision support"},"reference_resolution":{"displayed":61,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":39,"verified_exact":1,"verified_fuzzy":21},"total_outbound_references":61},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 61 of 61 outbound references and 1 inbound Pith citation observation for arXiv:2507.18143."}