{"as_of":"2026-08-10T04:06:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:bfec4c92f9bef919071e2db876c7c3cde7186ab8caa87706c0f3b58904e69914","coverage":[{"denominator":67,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":67,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-09T16:10:24.939025Z","state":"measured"},{"denominator":68,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":68,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T15:28:47.559357Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-06T15:28:49.752882Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"cited_work":{"arxiv_id":"2502.01243","doi":null,"metadata_source":"pith","pith_arxiv_id":"2502.01243","snapshot_observed_at":"2026-08-06T15:28:49.752882Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","venue":"cs.CL","work_id":"0dcbc693-5039-44bd-abcf-d5d394440e21","year":2025},"citing_paper":{"arxiv_id":"2507.15717","last_updated":"2025-07-21T15:27:32Z","snapshot_observed_at":"2026-08-09T18:16:25.173548Z","submitted_at":"2025-07-21T15:27:32Z","title":"BEnchmarking LLMs for Ophthalmology (BELO) for Ophthalmological Knowledge and Reasoning","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T15:28:47.559357Z"},"links":{"cited_paper":"/paper/2502.01243","citing_paper":"/paper/2507.15717"},"observation_digest":"sha256:7828f7b047eb0f76964d368bb2809bab32ee5b5fee6193c2643bd40ee6148dad","observation_id":"71bd979c-52ef-432c-96cc-8e3fa3c7fbc4","resolution":{"observed_at":"2026-08-06T15:28:49.862916Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2502.01243/citation-record","integrity":"/paper/2502.01243/integrity","json":"/paper/2502.01243/citation-record.json","paper":"/paper/2502.01243"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.755672Z","title":"Language models are few-shot learners","venue":null,"work_id":"e12c66d8-c8ff-4c4c-9849-95c06622b7bd","year":1901},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.678965Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:9aeaa4e16ca618e69a44c2f3cc65f23118dd1af7c086937acd2d1d9a970dd434","observation_id":"7e99fa3b-888c-4a5c-90f2-199ff7fc5cd1","resolution":{"observed_at":"2026-08-09T16:10:25.760346Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.743318Z","title":"Gpt-4 technical report","venue":null,"work_id":"8c23b024-6f81-48f6-9463-0e9ea1800f97","year":2023},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.683514Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:9258c4e21ef1bb2ab9468fd68018bfd69343567910118f1d8e9e88c8d2a3e67e","observation_id":"1de9be11-abb6-45fd-949e-25ebf9fae69d","resolution":{"observed_at":"2026-08-09T16:10:25.747043Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.731755Z","title":"Large language models encode clinical knowledge","venue":null,"work_id":"f0ad5b36-cd48-4260-a10f-c82fd6a3a293","year":2023},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.688490Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:3d9791a464dca9220837e2b445b6e585c51ac9501be00b301706112a24a26762","observation_id":"a41e2ddc-1d59-44ec-bd1e-246b0d7f4905","resolution":{"observed_at":"2026-08-09T16:10:25.735523Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.720781Z","title":"Empowering biomedical discovery with ai agents","venue":null,"work_id":"f817a27c-a2b2-44dc-bad3-9861f83038dc","year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.692316Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:34f122ee459925db058488d6123a242aae031b430c5e2628b27baf063b664a7c","observation_id":"0d3b57c8-360b-4980-8481-7f8ab45a301a","resolution":{"observed_at":"2026-08-09T16:10:25.724649Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.01769","last_updated":"2024-11-21T23:39:12Z","snapshot_observed_at":"2026-08-06T16:39:54.371324Z","submitted_at":"2024-05-02T22:43:02Z","title":"A Survey on Large Language Models for Critical Societal Domains: Finance, Healthcare, and Law","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.01769","snapshot_observed_at":"2026-08-09T16:10:24.696498Z","title":"A survey on large language models for critical societal domains: finance, healthcare, and law","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.696498Z"},"links":{"cited_paper":"/paper/2405.01769","citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:89873ce46f70364f040e03b2b8de36434a0fa4cc90572af79013155afb056f0b","observation_id":"e97a4fec-78c5-490b-9977-a1b61d53dbdd","resolution":{"observed_at":"2026-08-09T16:10:24.696498Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.708285Z","title":"Integrated image- based deep learning and language models for primary diabetes care","venue":null,"work_id":"fb864147-baf8-494f-8824-86c8a42baf65","year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.700872Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:265a338b02e7a5a8e4e70e7ca112d5ff6ce5691bbccb4a7253ca71d64ae4fa44","observation_id":"6e62f7f1-09b0-4069-89fb-fe5ad06c5a24","resolution":{"observed_at":"2026-08-09T16:10:25.712403Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.696351Z","title":"Evaluating large language models on medical evidence summarization","venue":null,"work_id":"4b340e6c-b077-4c43-80f2-26ea57256519","year":2023},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.705174Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:936c5146b76794325d231c59ee21dfa7a3e892cd31326a68f8cf27b51dc00512","observation_id":"8d44da20-e9a1-42d2-9b23-0aee28cfa1b5","resolution":{"observed_at":"2026-08-09T16:10:25.700623Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.684360Z","title":"Adapted large language models can outperform medical experts in clinical text summarization","venue":null,"work_id":"0ea62a4a-e887-4fe5-ab08-c8fff34c3910","year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.708747Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:5b830c294393f7e2076b32a8a88b9c90f3d0c061434b1d46519e08c8209c6c04","observation_id":"bc5c8d60-4222-432e-b65e-4f1038fa2f4b","resolution":{"observed_at":"2026-08-09T16:10:25.688533Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.671750Z","title":"A strategy for cost-effective large language model use at health system-scale","venue":null,"work_id":"a40df2dd-710b-4a51-a907-8d4004c0931f","year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.712213Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:140f381e8c7679da623afb7998d4dd99bab12b8b3610e16b7c7d8194f76a5478","observation_id":"5fe34d09-1522-47b6-80be-77a2e93fb158","resolution":{"observed_at":"2026-08-09T16:10:25.676461Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.660092Z","title":"Matching patients to clinical trials with large language models","venue":null,"work_id":"efe347f4-f7c5-495c-a129-ecb6814fd454","year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.716163Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:9a8f7911f262b5c77b855f66e884c4bda01c05e693604eab1fa05aece7e63c39","observation_id":"182b487d-c709-4e94-8f91-c865bc0549c7","resolution":{"observed_at":"2026-08-09T16:10:25.663887Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.648481Z","title":"Scaling clinical trial matching using large language models: a case study in oncology","venue":null,"work_id":"734aa46c-0c31-4d2f-9954-d16009b595f7","year":2023},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.720054Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:fced54607591c0f2384b2a73104d9934adf3e04095abad7ae47f27e8df6b689b","observation_id":"a4da407d-a80b-4065-b7fb-d86764577b6f","resolution":{"observed_at":"2026-08-09T16:10:25.652381Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.636071Z","title":"Chatgpt and other large language models are double-edged swords, 2023","venue":null,"work_id":"53f02644-76ed-41b4-8454-df5355babded","year":2023},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.723503Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:2b78bdf9bdd7a4e17f6a94cd001674a46e716971b257c7d9a48fea61deddecbd","observation_id":"da2f6e5f-dd5a-4797-8de1-143aa2b10568","resolution":{"observed_at":"2026-08-09T16:10:25.640220Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.624676Z","title":"Ethics of large language models in medicine and medical research","venue":null,"work_id":"5e9ed5fa-4278-4939-9273-511c7c80e7e5","year":2023},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.727096Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:6350126d60506b3ce23e9230449ff82fb26ffd98e4e7239dd00d62077f0903e9","observation_id":"bf522576-4202-40ae-b728-d052f4c49912","resolution":{"observed_at":"2026-08-09T16:10:25.628493Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.612397Z","title":"The ethics of chatgpt in medicine and healthcare: a systematic review on large language models (llms)","venue":null,"work_id":"6195a5bc-c571-4e32-b176-eee5f02c2f51","year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.730761Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:f2bcd7255c85200c28e72ee0c0ca5fba8468e2668038ef4ecdab560808a69df3","observation_id":"c874e2de-d98c-4048-bf06-dc518f7af110","resolution":{"observed_at":"2026-08-09T16:10:25.616731Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.600462Z","title":"Clinical large language models with misplaced focus","venue":null,"work_id":"266185dc-51ca-419e-a38d-35d22b81ae44","year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.734504Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:7923d76e27d316e1bb0984a75f2c343c308583f6e317aa0b9fecb0e64cbdcc49","observation_id":"a4f1d673-17c7-49cb-a5a8-2ab014bf55a3","resolution":{"observed_at":"2026-08-09T16:10:25.604718Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.588660Z","title":"Medmcqa: A large-scale multi-subject multi-choice dataset for medical domain question answering","venue":null,"work_id":"24a9a204-bd5d-4265-87da-624543bef875","year":2022},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.737963Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:fcad738372e7174f391f72b826a365a792482ebe361409134e59f1558846483d","observation_id":"d3b74ea7-6568-4de2-9581-ade055a3e330","resolution":{"observed_at":"2026-08-09T16:10:25.593609Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12036","last_updated":"2024-06-30T15:12:10Z","snapshot_observed_at":"2026-07-06T18:32:32.228906Z","submitted_at":"2024-06-17T19:07:21Z","title":"MedCalc-Bench: Evaluating Large Language Models for Medical Calculations","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12036","snapshot_observed_at":"2026-08-09T16:10:24.741545Z","title":"Medcalc- bench: Evaluating large language models for medical calculations","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.741545Z"},"links":{"cited_paper":"/paper/2406.12036","citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:99d4d7076f17986c832f2043961c527f4b4120a65764d98ea3c8ab2a263b7602","observation_id":"b76d961a-4776-4dd0-b14d-602468740e1c","resolution":{"observed_at":"2026-08-09T16:10:24.741545Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.14490","last_updated":"2024-01-25T19:57:00Z","snapshot_observed_at":"2026-08-08T05:12:21.493140Z","submitted_at":"2024-01-25T19:57:00Z","title":"LongHealth: A Question Answering Benchmark with Long Clinical Documents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.14490","snapshot_observed_at":"2026-08-09T16:10:24.745537Z","title":"Longhealth: A question answering benchmark with long clinical documents","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.745537Z"},"links":{"cited_paper":"/paper/2401.14490","citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:3220890007c8896caddcb39bd864eed284ea9cf57aba5fd09f855e6d13d9e3a8","observation_id":"a8cc993e-f3a7-42a1-b109-dd93db1cb612","resolution":{"observed_at":"2026-08-09T16:10:24.745537Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.08087","last_updated":"2022-03-07T09:14:20Z","snapshot_observed_at":"2026-08-01T20:36:18.568223Z","submitted_at":"2021-06-15T12:25:30Z","title":"CBLUE: A Chinese Biomedical Language Understanding Evaluation Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.08087","snapshot_observed_at":"2026-08-09T16:10:24.749724Z","title":"Cblue: A chinese biomedical language understanding evaluation benchmark.arXiv preprint arXiv:2106.08087, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.749724Z"},"links":{"cited_paper":"/paper/2106.08087","citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:ec7cb94b399ffb31089c0d8713a455e0748aa65481680394fa04f24e48db1b4a","observation_id":"923368a5-8d70-4a75-a643-63a5321f5411","resolution":{"observed_at":"2026-08-09T16:10:24.749724Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.577084Z","title":"case of the month","venue":null,"work_id":"b145f74a-1876-4f3b-9c0a-66da92d06bfd","year":2023},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.753613Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:82602277d674ba1861242c30b29634b36561b2dc1bb9c201c60af4bf0662c50b","observation_id":"52a7410b-c299-43fc-8f7b-65288a1214c1","resolution":{"observed_at":"2026-08-09T16:10:25.580838Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.565024Z","title":"Accuracy and reliability of chatbot responses to physician questions","venue":null,"work_id":"fb29a252-41b5-48b6-9bc3-054f43c53c02","year":2023},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.757046Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:9206f05941c57587f419ef28b5fcd3d4acae9fbddbef5256e01ac96b19d9e833","observation_id":"08b4cb75-e5d1-4176-a539-5258376ce987","resolution":{"observed_at":"2026-08-09T16:10:25.569108Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.552248Z","title":"Capabilities of gpt-4 in ophthalmology: an analysis of model entropy and progress towards human-level medical question answering","venue":null,"work_id":"5607c2fc-7fe6-436e-a34f-3d14b8cf9abe","year":2023},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.760571Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:a97ee9efbeedf5854855f8dcebacc7d48552d5473c70cdfc5c36d3fc8a2f9f0b","observation_id":"d295919a-1296-4748-8b26-6cdc5da31ebe","resolution":{"observed_at":"2026-08-09T16:10:25.556229Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.540389Z","title":"Comparison of ophthalmologist and large language model chatbot responses to online patient eye care questions","venue":null,"work_id":"5ad72ade-cb72-4637-b399-72c2fa9629a5","year":2023},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.763952Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:9566e7c62e58e1487e779536b35d221a1737ee357faa89c127bedfa7d082f442","observation_id":"0f6d3176-eabc-423b-85c6-d0d35b2d3257","resolution":{"observed_at":"2026-08-09T16:10:25.544626Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.528815Z","title":"Eval- uation and mitigation of the limitations of large language models in clinical decision-making","venue":null,"work_id":"7f2cf14c-9652-42d6-8ea9-7f3c008cd26b","year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.768000Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:daa5465e50d487aed14a290bd16037ee0f7aa03de6b19a69fe8564cdefcf84dc","observation_id":"2a944e9d-4e16-4d65-b16e-455948925008","resolution":{"observed_at":"2026-08-09T16:10:25.532999Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.516814Z","title":"Assessing the utility of chatgpt throughout the entire clinical workflow: development and usability study.J","venue":null,"work_id":"879da464-b37a-464b-9fcb-b785aa11e948","year":2023},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.771483Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:49d667d3e305816da22b8821ea2429cffe8d9f78f2337a2acb5b9ef40f052860","observation_id":"aa21e586-7bd3-457d-8355-710588fe5e0d","resolution":{"observed_at":"2026-08-09T16:10:25.520990Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.504617Z","title":"Benchmarking large language models’ performances for myopia care: a comparative analysis of chatgpt- 3.5, chatgpt-4.0, and google bard","venue":null,"work_id":"552b45b2-556d-45b9-94fa-abb837c560d1","year":2023},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.775135Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:2db75ebfb4ae39c30cccb39ce10b9762c51da9eaf9b7274ba19cd167197e5ad8","observation_id":"3abfeb09-357f-4ed7-a9ed-cd5a0c6b1ec4","resolution":{"observed_at":"2026-08-09T16:10:25.508808Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.491929Z","title":null,"venue":null,"work_id":"e61acf88-8959-4765-905c-1649bed58553","year":2023},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.779033Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:a783d981e64542c1ce1a0f05bf7cefe60f45dff8c9bacd7f43a507e5183bdce0","observation_id":"10b8a1cc-80fb-48e3-bee7-a67abab142e6","resolution":{"observed_at":"2026-08-09T16:10:25.496109Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.478604Z","title":"Performance of large language models on medical oncology examination questions","venue":null,"work_id":"af669f1b-d591-4f8e-a8be-8d4c44e4fef2","year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.782880Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:df43b43d83203b5bf705eebdff7c4877cb7ac014081cc5bca0eacfbc06ee4259","observation_id":"19d09383-f60a-4b06-91c2-6e860c2edb0c","resolution":{"observed_at":"2026-08-09T16:10:25.483260Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.466143Z","title":"Assessment of a large lan- guage model’s responses to questions and cases about glaucoma and retina management","venue":null,"work_id":"23dedf0b-939b-4c89-bfb2-cbc3bb783cc2","year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.786529Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:15b5e314f49efaf7511da0c3fb918b1ddd78f5c63bf813cac38d2b5ba5c537c6","observation_id":"04624a96-4206-4d55-9dfa-f084ee9eb189","resolution":{"observed_at":"2026-08-09T16:10:25.471017Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2009.13081","last_updated":"2020-09-28T05:07:51Z","snapshot_observed_at":"2026-08-06T03:17:52.286711Z","submitted_at":"2020-09-28T05:07:51Z","title":"What Disease does this Patient Have? A Large-scale Open Domain Question Answering Dataset from Medical Exams","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.13081","snapshot_observed_at":"2026-08-09T16:10:24.790099Z","title":"What disease does this patient have? A large-scale open domain question answering dataset from medical exams","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.790099Z"},"links":{"cited_paper":"/paper/2009.13081","citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:63f8cefc466265966a64c4c35f9446168cb1535a4079c58d8c0bb0801c4ccf44","observation_id":"5d08a686-870b-49d0-8346-787cd400db66","resolution":{"observed_at":"2026-08-09T16:10:24.790099Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.454160Z","title":"Medmcqa: A large-scale multi-subject multi-choice dataset for medical domain question answering","venue":null,"work_id":"3e9c036c-8b81-433d-86c5-93288484190c","year":2022},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.793887Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:ddf09ad994d19acc4c2ea8f6fd337cbfa529e950b799a9905aa49bd10cb278be","observation_id":"ceff7c40-0281-4790-aa80-bf67584a4be2","resolution":{"observed_at":"2026-08-09T16:10:25.457888Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.442916Z","title":"Benchmarking large language models on cmexam-a comprehensive chinese medical exam dataset","venue":null,"work_id":"c35f482f-b837-4e3b-b67e-40a5036c5095","year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.798044Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:df35c797f826ade69f2148e002df06b87c6083b9ad139cbddf465b38cbf33087","observation_id":"2311a256-556e-4f71-98ca-186de61b45c2","resolution":{"observed_at":"2026-08-09T16:10:25.447084Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.432010Z","title":"Medbench: A large- scale chinese benchmark for evaluating medical large language models","venue":null,"work_id":"d21c7800-a5fc-4048-9325-2166b81d00c6","year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.801431Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:a85fd647a948ce65bc8e08b60fe51cb309d728ea3e9362d8708743727a3ffe46","observation_id":"617b143a-964f-4a98-9b4a-5eacc8cc4831","resolution":{"observed_at":"2026-08-09T16:10:25.435761Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.13317","last_updated":"2024-09-20T08:25:16Z","snapshot_observed_at":"2026-07-06T19:18:36.400938Z","submitted_at":"2024-09-20T08:25:16Z","title":"JMedBench: A Benchmark for Evaluating Japanese Biomedical Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.13317","snapshot_observed_at":"2026-08-09T16:10:24.805634Z","title":"Jmedbench: A benchmark for evaluating japanese biomedical large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.805634Z"},"links":{"cited_paper":"/paper/2409.13317","citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:227e2e6402daec417d01dd1e750fd47d4d1f17f401f214fe73d81b7be35abfc7","observation_id":"f673d7e5-430d-48c9-94d8-88119444602f","resolution":{"observed_at":"2026-08-09T16:10:24.805634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.420567Z","title":null,"venue":null,"work_id":"17de84c6-dc43-4b02-93e5-de6f88c5d4ea","year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.809987Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:2e9a02bed5df0ed433524299d49f0c58fd0c076029568c745674799260f06ce2","observation_id":"bc0ad3c2-dcb6-41d1-9372-a96d7608da06","resolution":{"observed_at":"2026-08-09T16:10:25.424731Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01126","last_updated":"2024-06-03T09:11:13Z","snapshot_observed_at":"2026-07-06T18:24:17.711867Z","submitted_at":"2024-06-03T09:11:13Z","title":"TCMBench: A Comprehensive Benchmark for Evaluating Large Language Models in Traditional Chinese Medicine","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01126","snapshot_observed_at":"2026-08-09T16:10:24.814407Z","title":"Tcmbench: A comprehensive benchmark for evaluating large language models in traditional chinese medicine","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.814407Z"},"links":{"cited_paper":"/paper/2406.01126","citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:81dfd87ae18b9ba90581ef44c3c665a9902cd96cb891bffa260008ef67ed2936","observation_id":"c57837f2-3a88-4656-b7a3-39b439c99c3a","resolution":{"observed_at":"2026-08-09T16:10:24.814407Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.12405","last_updated":"2024-10-16T09:38:13Z","snapshot_observed_at":"2026-07-06T19:34:27.813248Z","submitted_at":"2024-10-16T09:38:13Z","title":"ProSA: Assessing and Understanding the Prompt Sensitivity of LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.12405","snapshot_observed_at":"2026-08-09T16:10:24.818576Z","title":"Prosa: Assessing and understanding the prompt sensitivity of llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.818576Z"},"links":{"cited_paper":"/paper/2410.12405","citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:a382332d68e3fa75abf68c58b31e457c46326cdf7481ce981c38c31320c1e004","observation_id":"d482c786-74a1-4a91-b16e-9883d91b5654","resolution":{"observed_at":"2026-08-09T16:10:24.818576Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15938","last_updated":"2024-05-31T17:49:03Z","snapshot_observed_at":"2026-08-03T17:29:17.531819Z","submitted_at":"2024-02-24T23:54:41Z","title":"Generalization or Memorization: Data Contamination and Trustworthy Evaluation for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.15938","snapshot_observed_at":"2026-08-09T16:10:24.822355Z","title":"Generalization or memorization: Data contamination and trustworthy evaluation for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.822355Z"},"links":{"cited_paper":"/paper/2402.15938","citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:a57e6363e0a80abac4a5d38a01c891ce43d954becc121d250da2bafb7f033382","observation_id":"2a71d522-7843-4352-9b75-b8e446b01855","resolution":{"observed_at":"2026-08-09T16:10:24.822355Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.407017Z","title":"Promptcblue: A chinese prompt tuning benchmark for the medical domain","venue":null,"work_id":"5e29bfd3-eef3-46da-b216-6435900d3e62","year":2023},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.826194Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:4e56deccd7406f2090e644a9fca596d67be62c590f88c79a6e9a4c3b06d10577","observation_id":"e820c545-33ae-423f-92a6-4eb234ebc7d3","resolution":{"observed_at":"2026-08-09T16:10:25.411148Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.08833","last_updated":"2024-04-04T15:16:57Z","snapshot_observed_at":"2026-07-06T16:07:10.648879Z","submitted_at":"2023-08-17T07:51:23Z","title":"CMB: A Comprehensive Medical Benchmark in Chinese","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.08833","snapshot_observed_at":"2026-08-09T16:10:24.829663Z","title":"Cmb: A comprehensive medical benchmark in chinese","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.829663Z"},"links":{"cited_paper":"/paper/2308.08833","citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:f00748db543951796d75a657a5b3c5e38e632de510f82de3552cd6beb56f4a57","observation_id":"799dc24d-aa5c-4314-b38e-520bde0bade2","resolution":{"observed_at":"2026-08-09T16:10:24.829663Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.11239","last_updated":"2024-10-02T11:07:14Z","snapshot_observed_at":"2026-08-03T08:44:22.056144Z","submitted_at":"2024-09-17T14:40:02Z","title":"LLM-as-a-Judge & Reward Model: What They Can and Cannot Do","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.11239","snapshot_observed_at":"2026-08-09T16:10:24.833277Z","title":"Llm-as-a-judge & reward model: What they can and cannot do","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.833277Z"},"links":{"cited_paper":"/paper/2409.11239","citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:e8f270fb55bf7500194bb244941631330cfbe59efebc18c6ec96ed32aa3c6a56","observation_id":"542fd470-72f3-4c92-bc67-fa278e2454c2","resolution":{"observed_at":"2026-08-09T16:10:24.833277Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.16256","last_updated":"2024-10-21T17:56:51Z","snapshot_observed_at":"2026-07-06T19:37:16.203681Z","submitted_at":"2024-10-21T17:56:51Z","title":"CompassJudger-1: All-in-one Judge Model Helps Model Evaluation and Evolution","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.16256","snapshot_observed_at":"2026-08-09T16:10:24.837215Z","title":"Compassjudger- 1: All-in-one judge model helps model evaluation and evolution","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.837215Z"},"links":{"cited_paper":"/paper/2410.16256","citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:f5413232157621deb69546cf82b9bf3a647e691110ba2d488c053d960c867516","observation_id":"1f0190b6-7b45-4a20-8858-e051a840d010","resolution":{"observed_at":"2026-08-09T16:10:24.837215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.393629Z","title":null,"venue":null,"work_id":"792e64f4-2ec7-45aa-b221-6e65d8ff2769","year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.840870Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:d9488ad6a5a56c14044a7fa68c014443b012608bb47902f28b93629a19e7064e","observation_id":"0238ef16-b694-4f66-b618-6eee003b9da5","resolution":{"observed_at":"2026-08-09T16:10:25.398747Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.06287","last_updated":"2025-02-28T07:54:16Z","snapshot_observed_at":"2026-08-01T06:27:54.806036Z","submitted_at":"2024-12-09T08:19:28Z","title":"PediaBench: A Comprehensive Chinese Pediatric Dataset for Benchmarking Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.06287","snapshot_observed_at":"2026-08-09T16:10:24.844582Z","title":"Pedi- abench: A comprehensive chinese pediatric dataset for benchmarking large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.844582Z"},"links":{"cited_paper":"/paper/2412.06287","citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:e642f0de7c0f5931a138004bff58e075d91e6551eb4d9ec224e2e94f79bf9dfd","observation_id":"0332e790-72a3-4434-ac38-195cd9d7d6db","resolution":{"observed_at":"2026-08-09T16:10:24.844582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.377937Z","title":"Diagnostic reasoning prompts reveal the potential for large language model interpretability in medicine.NPJ Digital Medicine, 7(1):20, 2024","venue":null,"work_id":"bbe911d5-49c9-46d8-9114-f4dd5c42fffe","year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.848907Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:527e40c9bdbbe9bfee977913447379247f913fc8589f05dd3298c0b145cb9db3","observation_id":"29c9148e-3065-47be-b73e-47d86c0a9184","resolution":{"observed_at":"2026-08-09T16:10:25.383395Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.364398Z","title":"Large language models and their impact in ophthalmology","venue":null,"work_id":"c2af21b5-0989-4368-9712-7656485d4ec5","year":2023},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.852615Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:ccb36e7b20d96d3f76ede67650e083238d5254c2d7ca6039acf9a992b3afbe2c","observation_id":"a03514cf-7089-4836-a45a-84fffe257f78","resolution":{"observed_at":"2026-08-09T16:10:25.368559Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.352760Z","title":"How i won singapore’s gpt-4 prompt engineering competition","venue":null,"work_id":"fa3795eb-17ba-4af7-8739-b62d16532f4f","year":2023},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.856125Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:0ce25160b840acf5db18315e20ae6881af1c949a808961a57e391f24a0abffbb","observation_id":"95626846-fcc5-4360-afec-46667e25a53b","resolution":{"observed_at":"2026-08-09T16:10:25.356894Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.340994Z","title":"Alignbench: Benchmarking chinese alignment of large language models","venue":null,"work_id":"530b13ca-42a2-4050-ba3f-52e9444f36d2","year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.859553Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:e35d93eb7a565b6dea384922634bef5f57602363a23130d1889e8589f44b20e1","observation_id":"e2adcad7-6b1f-4bca-811c-b457c87c2e4e","resolution":{"observed_at":"2026-08-09T16:10:25.344774Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.10305","last_updated":"2025-04-17T08:34:42Z","snapshot_observed_at":"2026-08-06T06:52:14.558389Z","submitted_at":"2023-09-19T04:13:22Z","title":"Baichuan 2: Open Large-scale Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.10305","snapshot_observed_at":"2026-08-09T16:10:24.863235Z","title":"Baichuan 2: Open large-scale language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.863235Z"},"links":{"cited_paper":"/paper/2309.10305","citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:fae62f1fd73e98b25cbbb7a6fffdb27116297ebaca49074fe34e6b216c542347","observation_id":"7867d21b-fc2c-4486-8ce2-cd684a6d88ef","resolution":{"observed_at":"2026-08-09T16:10:24.863235Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.15075","last_updated":"2023-05-24T11:56:01Z","snapshot_observed_at":"2026-08-07T23:04:52.692705Z","submitted_at":"2023-05-24T11:56:01Z","title":"HuatuoGPT, towards Taming Language Model to Be a Doctor","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.15075","snapshot_observed_at":"2026-08-09T16:10:24.867232Z","title":"Huatuogpt, towards taming language model to be a doctor","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.867232Z"},"links":{"cited_paper":"/paper/2305.15075","citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:7e48d413b9996336577fafd2db6d518d00ccdcbd4512c991bbf20712d2228984","observation_id":"c66560ef-6cd1-4834-9e99-49d37cb6c9f6","resolution":{"observed_at":"2026-08-09T16:10:24.867232Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.18925","last_updated":"2024-12-25T15:12:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-25T15:12:34Z","title":"HuatuoGPT-o1, Towards Medical Complex Reasoning with LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.18925","snapshot_observed_at":"2026-08-09T16:10:24.871252Z","title":"Huatuogpt-o1, towards medical complex reasoning with llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.871252Z"},"links":{"cited_paper":"/paper/2412.18925","citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:ea68d8618211b837daeb69024540193e832e2b99577450e4f2d180791c4aeb32","observation_id":"49c815b8-5cbb-4294-b0d1-7533aea3b265","resolution":{"observed_at":"2026-08-09T16:10:24.871252Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15124","last_updated":"2025-04-14T22:39:09Z","snapshot_observed_at":"2026-08-08T12:58:42.430328Z","submitted_at":"2024-11-22T18:44:04Z","title":"Tulu 3: Pushing Frontiers in Open Language Model Post-Training","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15124","snapshot_observed_at":"2026-08-09T16:10:24.875085Z","title":"T \\\" ulu 3: Pushing frontiers in open language model post-training","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.875085Z"},"links":{"cited_paper":"/paper/2411.15124","citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:70acdc06dc6d8a7ad3042f78725f89ff9d240656ac520befbe72c55453f59300","observation_id":"38af74d0-459f-4dcf-8bda-8b5ab6c35937","resolution":{"observed_at":"2026-08-09T16:10:24.875085Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-09T16:10:24.879671Z","title":"The llama 3 herd of models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.879671Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:f35f6e466c17668b762d2988533c61076e0e7b7dfd128efd242fa69fbf34507f","observation_id":"e6d5d9ff-993a-41a4-b20b-8023d6e0f0cd","resolution":{"observed_at":"2026-08-09T16:10:24.879671Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06825","last_updated":"2023-10-10T17:54:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-10T17:54:58Z","title":"Mistral 7B","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06825","snapshot_observed_at":"2026-08-09T16:10:24.884551Z","title":"Mistral 7b","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.884551Z"},"links":{"cited_paper":"/paper/2310.06825","citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:59875a7c2a76615c7181b191227d5c0181c082648a05260e585fc5e4a1a4579d","observation_id":"04f9f1ef-dd9e-4ef3-b304-73b2124f4664","resolution":{"observed_at":"2026-08-09T16:10:24.884551Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.328291Z","title":"Pulse: Pretrained and unified language service engine","venue":null,"work_id":"c7ba6204-bb0c-4b69-8f17-41df49ed1c95","year":2023},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.888627Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:bc956d20e6a8a33233e0320c04ee7fb0499bdb3db8b35a6346f7b86168c97239","observation_id":"92b88e27-3741-40ac-a3f7-066036cc0196","resolution":{"observed_at":"2026-08-09T16:10:25.333032Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.14219","last_updated":"2024-08-30T21:17:17Z","snapshot_observed_at":"2026-07-06T18:03:47.096406Z","submitted_at":"2024-04-22T14:32:33Z","title":"Phi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.14219","snapshot_observed_at":"2026-08-09T16:10:24.892214Z","title":"Phi-3 technical report: A highly capable language model locally on your phone","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.892214Z"},"links":{"cited_paper":"/paper/2404.14219","citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:529e5342ae7fd7744375e3fed4d16538c51744a942f7eec4b30df2161c24c5ed","observation_id":"0ffc44dc-3f9e-4072-ad93-5047c867595e","resolution":{"observed_at":"2026-08-09T16:10:24.892214Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-09T16:10:24.896839Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.896839Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:6518d9fd085d47ef6a9e9da9f9f42ec55c4ddf4885c968ffaeabd94bf42cc6e4","observation_id":"5089a42f-448a-4ab3-b802-ee31256a8587","resolution":{"observed_at":"2026-08-09T16:10:24.896839Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:25.315257Z","title":"Sunsimiao: Chinese medicine llm","venue":null,"work_id":"f37bd66d-b261-4200-8ca2-7789cb19a34b","year":2023},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.901000Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:9d79b673274e68f8fec251b5cb2260522d62d9bfd73637171dd91081bab38edf","observation_id":"2ba5ce7a-fd94-459e-8e69-a54ec54bef7f","resolution":{"observed_at":"2026-08-09T16:10:25.320401Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04652","last_updated":"2025-01-21T10:12:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-07T16:52:49Z","title":"Yi: Open Foundation Models by 01.AI","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04652","snapshot_observed_at":"2026-08-09T16:10:24.906158Z","title":"Yi: Open foundation models by 01","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.906158Z"},"links":{"cited_paper":"/paper/2403.04652","citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:704a61b3c5f5eb659f961f37053e5c48980cac3330f648d105982ced4aaffd07","observation_id":"d0c82835-991b-431f-be27-acdbc325c613","resolution":{"observed_at":"2026-08-09T16:10:24.906158Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.02954","last_updated":"2024-01-05T18:59:13Z","snapshot_observed_at":"2026-08-02T13:11:16.882565Z","submitted_at":"2024-01-05T18:59:13Z","title":"DeepSeek LLM: Scaling Open-Source Language Models with Longtermism","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.02954","snapshot_observed_at":"2026-08-09T16:10:24.910143Z","title":"Deepseek llm: Scaling open-source language models with longtermism","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.910143Z"},"links":{"cited_paper":"/paper/2401.02954","citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:c8338f2dcfa65948053fe8c342f04ae833f373a99e8d524079af88fc2ee04094","observation_id":"787bc067-66e8-493d-aa49-09e26bf4665c","resolution":{"observed_at":"2026-08-09T16:10:24.910143Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00118","last_updated":"2024-10-02T15:22:49Z","snapshot_observed_at":"2026-08-02T16:20:09.773989Z","submitted_at":"2024-07-31T19:13:07Z","title":"Gemma 2: Improving Open Language Models at a Practical Size","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00118","snapshot_observed_at":"2026-08-09T16:10:24.914543Z","title":"Gemma 2: Improving open language models at a practical size","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.914543Z"},"links":{"cited_paper":"/paper/2408.00118","citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:db05f78be7b2ff87a55d0ed1043f14652aa18ba37ee92ddc392404dc53df7a08","observation_id":"63b8aace-8179-418c-a570-6e8ba37b641e","resolution":{"observed_at":"2026-08-09T16:10:24.914543Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T16:10:24.919428Z","title":"Granite 3.0 language models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.919428Z"},"links":{"citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:386315c8e365e735bd14d354ab913aea67ffb9328391123a5f6d58f4915f0d85","observation_id":"c39933b6-4b92-4ea4-a584-cef701fc2252","resolution":{"observed_at":"2026-08-09T16:10:24.919428Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12793","last_updated":"2024-07-30T03:58:11Z","snapshot_observed_at":"2026-08-07T13:56:34.167869Z","submitted_at":"2024-06-18T16:58:21Z","title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12793","snapshot_observed_at":"2026-08-09T16:10:24.923476Z","title":"Chatglm: A family of large language models from glm-130b to glm-4 all tools","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.923476Z"},"links":{"cited_paper":"/paper/2406.12793","citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:2f18a833637de76732e220b3ea14ae70b687c8aba5774c2884ff30c8a8b2f2d4","observation_id":"c080ce6c-7d1b-468a-801a-5a99007e2d3f","resolution":{"observed_at":"2026-08-09T16:10:24.923476Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17297","last_updated":"2024-03-26T00:53:24Z","snapshot_observed_at":"2026-08-02T11:10:24.263044Z","submitted_at":"2024-03-26T00:53:24Z","title":"InternLM2 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.17297","snapshot_observed_at":"2026-08-09T16:10:24.927331Z","title":"Internlm2 technical report","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.927331Z"},"links":{"cited_paper":"/paper/2403.17297","citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:7115042d4bdd5d6f41b245e0b5a4b41f87cffd8421d47e11cf9fc2eb7701b4e7","observation_id":"33d9a44c-321b-4024-a45e-3c6d7aea7a3f","resolution":{"observed_at":"2026-08-09T16:10:24.927331Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.04434","last_updated":"2024-06-19T06:04:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-05-07T15:56:43Z","title":"DeepSeek-V2: A Strong, Economical, and Efficient Mixture-of-Experts Language Model","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.04434","snapshot_observed_at":"2026-08-09T16:10:24.931071Z","title":"Deepseek-v2: A strong, economical, and efficient mixture-of-experts language model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.931071Z"},"links":{"cited_paper":"/paper/2405.04434","citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:3e1aaa0e59e6d2a87fac78ddf1490a4912efde51d7e27d053a1e0ff08480c463","observation_id":"7071e68a-a6a1-400c-b327-0df162bf3b31","resolution":{"observed_at":"2026-08-09T16:10:24.931071Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02265","last_updated":"2024-11-06T09:15:27Z","snapshot_observed_at":"2026-08-04T12:39:25.666793Z","submitted_at":"2024-11-04T16:56:26Z","title":"Hunyuan-Large: An Open-Source MoE Model with 52 Billion Activated Parameters by Tencent","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.02265","snapshot_observed_at":"2026-08-09T16:10:24.934912Z","title":"Hunyuan-large: An open-source moe model with 52 billion activated parameters by tencent","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.934912Z"},"links":{"cited_paper":"/paper/2411.02265","citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:6a747c7ca29be590d9b70f8453b346804b2544f2ffc94443e9fa9584585b1ca8","observation_id":"5a68a059-4596-4287-89b4-b991232b8963","resolution":{"observed_at":"2026-08-09T16:10:24.934912Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-08-09T16:10:24.939025Z","title":"Deepseek-v3 technical report","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-09T16:10:24.939025Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2502.01243"},"observation_digest":"sha256:8473391a75f8dd17544112a1e9f63d307df5d2d863493458c1bacae40ea5b520","observation_id":"9cda63fa-7b6c-4bb5-9f13-9021b92ca758","resolution":{"observed_at":"2026-08-09T16:10:24.939025Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2502.01243","last_updated":"2025-02-03T11:04:51Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-09T15:53:41.115212Z","submitted_at":"2025-02-03T11:04:51Z","title":"OphthBench: A Comprehensive Benchmark for Evaluating Large Language Models in Chinese Ophthalmology"},"reference_resolution":{"displayed":67,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":33,"verified_exact":0,"verified_fuzzy":34},"total_outbound_references":67},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 67 of 67 outbound references and 1 inbound Pith citation observation for arXiv:2502.01243."}