{"as_of":"2026-08-23T00:07:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:293baa0091b717d1f536313b075276cf76cc016b87f72c7905b24043ac587eb2","coverage":[{"denominator":71,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":71,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T16:16:49.312244Z","state":"measured"},{"denominator":71,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":71,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-22T06:32:14.747728+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2509.07135/citation-record","integrity":"/paper/2509.07135/integrity","json":"/paper/2509.07135/citation-record.json","paper":"/paper/2509.07135"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2005.14165","last_updated":"2020-07-22T19:47:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-05-28T17:29:03Z","title":"Language Models are Few-Shot Learners","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.14165","snapshot_observed_at":"2026-08-15T16:16:49.069536Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.069536Z"},"links":{"cited_paper":"/paper/2005.14165","citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:0a05d19212ab2c69cade1b735c3754a57705350f9c55cf8816a49bfe11a42b56","observation_id":"e65df800-a62f-4fd7-bcf7-10705eceaee8","resolution":{"observed_at":"2026-08-15T16:16:49.069536Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:50.051606Z","title":"Kasneci, K","venue":null,"work_id":"ac0f61f3-6cea-4db2-81e6-8cd5a19aed15","year":2023},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.073659Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:2177a7597d21ec3c353d9a1b6cda7c12fec1f889b5c4984c15a683ab532ab1db","observation_id":"0e6010d7-479f-4ed2-b91b-2236d9014be6","resolution":{"observed_at":"2026-08-15T16:16:50.054572Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:50.041283Z","title":"Baidoo-Anu, L","venue":null,"work_id":"f131cfda-908b-4833-82ad-140c26f7b3fe","year":2023},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.077500Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:14127762ca31152b5b6cd0a2b2d6bd8e878adc6f2ac6a4c6800e8b4e98604e1e","observation_id":"a2e3a83b-9b98-486e-82e4-58a45ea40cc5","resolution":{"observed_at":"2026-08-15T16:16:50.044766Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-20T16:26:42.002863Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-15T16:16:49.082371Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.082371Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:7132cde10feb3117a6c7f7249fda4a5ecf81dc006c45132121f24f27862f25a3","observation_id":"326e9dee-dbb8-4493-b0df-d5d650d6b192","resolution":{"observed_at":"2026-08-15T16:16:49.082371Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1905.00537","last_updated":"2020-02-13T00:28:00Z","snapshot_observed_at":"2026-08-17T22:08:45.245439Z","submitted_at":"2019-05-02T00:41:50Z","title":"SuperGLUE: A Stickier Benchmark for General-Purpose Language Understanding Systems","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1905.00537","snapshot_observed_at":"2026-08-15T16:16:49.086389Z","title":"Wang, et al., SuperGLUE: A Stickier Bench- mark for General-Purpose Language Understand- ing Systems, 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.086389Z"},"links":{"cited_paper":"/paper/1905.00537","citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:e108a52e850202023894b64f5bc22aede34280b61ec3403c676b348acd289032","observation_id":"204fa921-5871-42bf-9b11-5d7a813520c1","resolution":{"observed_at":"2026-08-15T16:16:49.086389Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:50.030016Z","title":"Hendrycks, C","venue":null,"work_id":"62620b15-e853-40f2-8b86-b6f60fea07d6","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.090279Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:944b585af07274d0213d38ddd1b6f53c37acb8057518f16b9bed0b275e645277","observation_id":"6667e3b0-de83-4fce-af9c-15949911fea6","resolution":{"observed_at":"2026-08-15T16:16:50.033443Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:50.018352Z","title":"Attanasio, P","venue":null,"work_id":"061cbe17-4b2b-403a-bf76-e804264c0076","year":2024},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.098386Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:9fd42d6761e8a06c79b7e7f1e8a65b0cbe354bdf4dc45ae350160b6a8941eeb4","observation_id":"c68c3aef-8b7e-4a85-bb0e-13549259717c","resolution":{"observed_at":"2026-08-15T16:16:50.022100Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:50.007217Z","title":"Moroni, S","venue":null,"work_id":"fc50e65a-29f7-491f-9f45-39c4f8499474","year":2024},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.101564Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:06675e7777b5799a0a5d492108673f8d0d88e71f1ec11b0bc00500a2fc6ebef5","observation_id":"419d4ade-231c-4b2d-a00d-bbd7f31016b2","resolution":{"observed_at":"2026-08-15T16:16:50.011347Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.989925Z","title":"Attanasio, P","venue":null,"work_id":"fa54144f-26c5-4900-b180-02586cad088e","year":2024},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.104902Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:9889bad08fd959b90ed1c5edab98b8ddecc7ad89b5099ab2e5287bbe3f96f2db","observation_id":"25e77af3-452b-4438-8f56-6fc147781370","resolution":{"observed_at":"2026-08-15T16:16:49.993573Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1804.07461","last_updated":"2019-02-22T23:53:34Z","snapshot_observed_at":"2026-08-16T09:50:11.319379Z","submitted_at":"2018-04-20T06:35:04Z","title":"GLUE: A Multi-Task Benchmark and Analysis Platform for Natural Language Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1804.07461","snapshot_observed_at":"2026-08-15T16:16:49.108419Z","title":"Wang, et al., GLUE: A Multi-Task Bench- mark and Analysis Platform for Natural Language Understanding, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.108419Z"},"links":{"cited_paper":"/paper/1804.07461","citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:85b6dffafd18d49b63d28ba2288158e30309c7bdeb4dc57447f7890688571898","observation_id":"b2f3cdf6-4b39-4373-b71a-8c7e00d947f8","resolution":{"observed_at":"2026-08-15T16:16:49.108419Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.112548Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.112548Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:3816db6e61e25a6fc7e01fb958cea91096fac81b02b41454ac22b01d61904ba0","observation_id":"9001183f-ac9c-4fb2-8b4a-25112fd11c0a","resolution":{"observed_at":"2026-08-15T16:16:49.112548Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1909.06146","last_updated":"2019-09-13T11:18:20Z","snapshot_observed_at":"2026-08-17T06:10:48.003173Z","submitted_at":"2019-09-13T11:18:20Z","title":"PubMedQA: A Dataset for Biomedical Research Question Answering","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.06146","snapshot_observed_at":"2026-08-15T16:16:49.116436Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.116436Z"},"links":{"cited_paper":"/paper/1909.06146","citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:771aa782bad0a1386fa8370e42eb707935ca251262e9b2ef71c1ba83b6dd5a8e","observation_id":"91c3b74b-c09a-46f5-a217-c44e7966071b","resolution":{"observed_at":"2026-08-15T16:16:49.116436Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.978424Z","title":null,"venue":null,"work_id":"2797f30b-0d58-46a8-b287-ac901b0d8610","year":2024},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.120056Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:b4d8966e638575a7e6e20f61fe12eb03b83497ec97747c68d710676b692d7d3a","observation_id":"7f42af82-5b87-4985-ade6-f85c4f43ce9b","resolution":{"observed_at":"2026-08-15T16:16:49.982195Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.968842Z","title":"Nentidis, K","venue":null,"work_id":"f146ce0f-94e4-44c3-b51c-4d1022336f2b","year":2023},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.123598Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:920fb1beda87621c0254fdd433d9b99e34b4bb00746c0a8d0465ca574bcec617","observation_id":"ace3c07e-32a7-4e8d-b114-26ce30898445","resolution":{"observed_at":"2026-08-15T16:16:49.972111Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.958527Z","title":"Rinaldi, J","venue":null,"work_id":"25195769-a31e-4092-96e6-64a0a700e0c5","year":2024},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.126443Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:b5ad2f16d6e50397914da6cd6afc9d4fbf324ab9c07cb5e1e94700e7268c5f38","observation_id":"23bc6808-d427-46ac-8787-18ece3a7d9bb","resolution":{"observed_at":"2026-08-15T16:16:49.962017Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.948990Z","title":"Casola, T","venue":null,"work_id":"b6112ecf-b6e2-4390-b4c7-6ee245ee7e99","year":2023},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.129599Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:e67e9e555e150303d6c45ab5cd5a9aa8130aa40bf6649dea04985639abbd02de","observation_id":"e89ac205-2c16-4b6f-913b-93f4e1898fb8","resolution":{"observed_at":"2026-08-15T16:16:49.952148Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.938578Z","title":"Altuna, G","venue":null,"work_id":"c787aeb1-bd9a-4339-a976-803ebfee37b6","year":2023},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.133420Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:5c490f670b5229c04cce90833a2d2e7814c50f22512de1c361dc741f5ec4295b","observation_id":"a6800399-d681-4e64-b362-6d9055d06838","resolution":{"observed_at":"2026-08-15T16:16:49.942246Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.928063Z","title":"Puccetti, M","venue":null,"work_id":"d6f1cd9a-b301-4597-b53a-a7399153b3cd","year":2025},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.136745Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:2d1c2c98328fd1e1ad56c730ab1617378ede20c27db3ec130e01157cfcc57451","observation_id":"af934513-ee0b-4917-ba29-535199513247","resolution":{"observed_at":"2026-08-15T16:16:49.931512Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2102.09690","last_updated":"2021-06-10T18:20:59Z","snapshot_observed_at":"2026-08-18T14:24:32.140370Z","submitted_at":"2021-02-19T00:23:59Z","title":"Calibrate Before Use: Improving Few-Shot Performance of Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2102.09690","snapshot_observed_at":"2026-08-15T16:16:49.139687Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.139687Z"},"links":{"cited_paper":"/paper/2102.09690","citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:e387e7611f83e7ad0178588bf49060209645b34dc7af9a0d765577b92606b051","observation_id":"8a896b5a-2d01-4cb1-b795-1bfb3a134c95","resolution":{"observed_at":"2026-08-15T16:16:49.139687Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.918705Z","title":"Wei, et al., Chain-of-Thought Prompting Elicits Reasoning in Large Language Models,","venue":null,"work_id":"139d2163-132e-45ec-a6cb-4d304e463baf","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.143562Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:e47cbedf93494d2c6bdc03324a22ad2ad849e92b400b3c389109a4bd20739815","observation_id":"69b2e6f1-05b6-48eb-aa13-7ce947d1ea26","resolution":{"observed_at":"2026-08-15T16:16:49.921769Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05229","last_updated":"2025-08-27T16:24:39Z","snapshot_observed_at":"2026-08-21T14:11:30.905906Z","submitted_at":"2024-10-07T17:36:37Z","title":"GSM-Symbolic: Understanding the Limitations of Mathematical Reasoning in Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05229","snapshot_observed_at":"2026-08-15T16:16:49.152173Z","title":"Mirzadeh, K","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.152173Z"},"links":{"cited_paper":"/paper/2410.05229","citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:72982cedcdacf1ced9f575d6728c7db6c55b9c493af3cda87c2b958104b9d561","observation_id":"d4e9987e-4e7b-403d-985c-f52aff34382e","resolution":{"observed_at":"2026-08-15T16:16:49.152173Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.909258Z","title":"Yang, et al., Qwen2.5 technical report,","venue":null,"work_id":"c7c2de3d-e26b-4890-9d99-7bfba6bb3e8b","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.155721Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:7c43aaf47666c2b5a45867e89194c9849bfb832c554ebded44aac34d62375cf6","observation_id":"6fbd38dc-624b-45bf-af5b-2ec106e795f7","resolution":{"observed_at":"2026-08-15T16:16:49.912340Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.08295","last_updated":"2024-04-16T12:52:47Z","snapshot_observed_at":"2026-08-03T03:29:01.959523Z","submitted_at":"2024-03-13T06:59:16Z","title":"Gemma: Open Models Based on Gemini Research and Technology","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.08295","snapshot_observed_at":"2026-08-15T16:16:49.164060Z","title":"Team, Gemma: Open Models Based on Gemini Research and Technology, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.164060Z"},"links":{"cited_paper":"/paper/2403.08295","citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:a55b4c36f50c0a0e608b088a32e13ae28566e2a8de34980238f18d7b2327ad4f","observation_id":"54b2dff8-c75a-4ea7-a8b9-d40a4a847185","resolution":{"observed_at":"2026-08-15T16:16:49.164060Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.900326Z","title":"Grattafiori, et al., The Llama 3 Herd of Models,","venue":null,"work_id":"b0d2b5b0-224a-467e-a3f1-30406bf46ccb","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.167514Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:9ff33331d0e9e3f659e15b40eabdc1d4cbf18f36a9959562068f50666b20be29","observation_id":"a406451c-01f9-401a-a742-9e474aca0834","resolution":{"observed_at":"2026-08-15T16:16:49.903468Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.14219","last_updated":"2024-08-30T21:17:17Z","snapshot_observed_at":"2026-08-17T03:25:04.404839Z","submitted_at":"2024-04-22T14:32:33Z","title":"Phi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.14219","snapshot_observed_at":"2026-08-15T16:16:49.174532Z","title":"Abdin, et al., Phi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.174532Z"},"links":{"cited_paper":"/paper/2404.14219","citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:d08763be444bee67bbe6eb6262d73b1f8a53a920f4ad17a9e4dc8cb58519d566","observation_id":"6020059f-a93d-45d9-a91f-736fc3a6a39d","resolution":{"observed_at":"2026-08-15T16:16:49.174532Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.890709Z","title":null,"venue":null,"work_id":"a97f7680-5920-4acd-8393-26ec68f905e0","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.178213Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:34f07793cc53093db8cd118e1d01f7f40583fe9f576da6cf0e588652bfed148a","observation_id":"e3bfd345-919a-43ef-a7bf-b711dfee23f8","resolution":{"observed_at":"2026-08-15T16:16:49.894241Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.184981Z","title":"Groeneveld, et al., OLMo: Accelerating the sci- ence of language models, in: L.-W","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.184981Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:ce79e8ce089f2e8456fdd9cbd9d3cc1e729b2dcc7ab64d2968f6b87ee5dc664b","observation_id":"70bba81e-5eb4-47ca-b173-59df023a23f9","resolution":{"observed_at":"2026-08-15T16:16:49.184981Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07827","last_updated":"2024-02-12T17:34:13Z","snapshot_observed_at":"2026-08-16T14:19:15.735686Z","submitted_at":"2024-02-12T17:34:13Z","title":"Aya Model: An Instruction Finetuned Open-Access Multilingual Language Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07827","snapshot_observed_at":"2026-08-15T16:16:49.188509Z","title":"Üstün, et al., Aya Model: An Instruc- tion Finetuned Open-Access Multilingual Lan- guage Model, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.188509Z"},"links":{"cited_paper":"/paper/2402.07827","citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:414051777658f8207deb1ec8b2e056e4bfc4d1248a3d1d4b132e772b4571e430","observation_id":"edd361d1-4dd9-4873-b94c-0b59175953ab","resolution":{"observed_at":"2026-08-15T16:16:49.188509Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.878364Z","title":"Orlando, L","venue":null,"work_id":"d234ac9e-774b-4fba-8278-f69284dcb4d6","year":2024},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.192297Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:31403d9c458255642dcb9bffed66f09cf4760d110389984d42c57a79a3787794","observation_id":"1b23dd12-0a49-42dc-802d-2af027aeef81","resolution":{"observed_at":"2026-08-15T16:16:49.882860Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.868452Z","title":"Tutti i bambini amano il gelato","venue":null,"work_id":"157f026f-483f-40dd-9b71-99c05ba993bf","year":2023},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.195766Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:fae8d0f48788aa7f76cf6dc245ad777d8351c65b0b46411c6f1443c55bbd316d","observation_id":"0b6a3a29-9f68-4616-a4b5-988c756e2d2a","resolution":{"observed_at":"2026-08-15T16:16:49.871689Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-08-15T12:33:55.451951Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-15T16:16:49.181295Z","title":"doi:10.48550/arXiv.2501.12948, arXiv:2501.12948 [cs]","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.181295Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:a4e7e19e5e76d5563b15eeb053b182dff1caa6d602acc00323c9066e4e58c0e1","observation_id":"ff3760e2-5407-4bdb-9036-30622c17c823","resolution":{"observed_at":"2026-08-15T16:16:49.181295Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.857987Z","title":"All children love ice cream","venue":null,"work_id":"93e227fd-c1da-4fc2-b669-b0b68c81ea1c","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.199048Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:a3d3ab012bd8963537452ed179dcf1c740073ae045430b8f7466e93c9bae6b94","observation_id":"6e278186-accb-4ab0-aded-4346cd79eeba","resolution":{"observed_at":"2026-08-15T16:16:49.861537Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.847409Z","title":"Logic Example Domanda:Se e solo se Giulia a luglio non va in vacanza in montagna, va poi in vacanza al mare ad agosto","venue":null,"work_id":"34889c26-08d6-4e77-a5c5-334bead5b810","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.203529Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:9013da4380405ab2a05282149ad402f70eb5669f033dfd14f909fa8be8e15b28","observation_id":"e807ec9b-b50e-44e1-bb32-b151c542ae4e","resolution":{"observed_at":"2026-08-15T16:16:49.850957Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.838010Z","title":"Carolina ha acquistato molte borse, dunque ha speso molti soldi","venue":null,"work_id":"6ecc61a1-4eaa-4fd7-8186-4c8940a5f9d8","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.206629Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:ec5596d22a3efe95f547d543a8f5df7eb6c29bb136aced5f80c935accdc70259","observation_id":"c801bf5b-9fc0-44a7-98a6-1562e4648bb8","resolution":{"observed_at":"2026-08-15T16:16:49.841115Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.829340Z","title":"Stasera non ha piovuto, dunque è andata in motorino","venue":null,"work_id":"a119fd3a-b8a9-4b65-9afb-0ec969515068","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.209773Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:b78ccc03be81d849456713fbc2d2c3194c72cc06f2d92b33be19fb149a215dc4","observation_id":"d916dd1e-97df-414f-9e65-696f04cd43ef","resolution":{"observed_at":"2026-08-15T16:16:49.832486Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.820917Z","title":"Ha già man- giato albicocche a pranzo, dunque a cena non mangia le fragole","venue":null,"work_id":"499e234d-bae7-4dd4-8b4a-4426b0a4109c","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.212754Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:e2be2c9fa3bc2b6e2911a41820dcf6243157accb6de95c4dc3c2bf3efc99ff77","observation_id":"e193d783-787d-499f-88d0-07bbdd8f1c22","resolution":{"observed_at":"2026-08-15T16:16:49.823889Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.812220Z","title":"Clara ha superato gli esami, dunque ha studi- ato molto","venue":null,"work_id":"5267c523-5466-4cdb-a973-5dce78a27cca","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.216582Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:913c7fce75db9b71a2d1dcb2ed1ab4d002317ec77f1f48a8fef115c89204b096","observation_id":"02ddee4a-1adf-4f59-bd04-7e046c8d81ac","resolution":{"observed_at":"2026-08-15T16:16:49.815477Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.802967Z","title":null,"venue":null,"work_id":"29d9c17f-51ad-4d58-9ca7-92cea610ec9a","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.219515Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:51251ac6b9ee45296bfcf06c44aa9792101ed774ab4508b080c52f64f1954035","observation_id":"89998033-3302-4271-87da-d6ce58e0dd1b","resolution":{"observed_at":"2026-08-15T16:16:49.805946Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.793362Z","title":"Carolina bought many bags, therefore she spent a lot of money","venue":null,"work_id":"c365c2a0-509c-45fb-ad1d-70ae2d99f497","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.222516Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:a5c71c2644f317fcca189f674bdd8a1a58d41d54f7b073032b32951b28630238","observation_id":"788372c4-1cc8-4420-8ddf-7e92d3bf499e","resolution":{"observed_at":"2026-08-15T16:16:49.796767Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.784614Z","title":"Tonight it did not rain, there- fore she went on her scooter","venue":null,"work_id":"2a1849bb-16d0-4ce8-b717-e83ea38cc189","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.225393Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:e4de5ae4cfc110ba11456cf08d2f4321905b4ec99ad78fec24f8c310655fac1f","observation_id":"79086ea3-cc52-4a2c-9711-0ce53e29b796","resolution":{"observed_at":"2026-08-15T16:16:49.787663Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.775237Z","title":"She already ate apricots for lunch, therefore she does not eat strawberries for dinner","venue":null,"work_id":"f434c625-2ec6-4840-b49b-b619f4e25578","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.228744Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:cb480b6c217df6d24d4ac0f2a3ed8e7ef48e2bb86def9602cec59aabf3ea78f5","observation_id":"c37903f5-366b-4eb0-a422-0eeab25fa248","resolution":{"observed_at":"2026-08-15T16:16:49.778524Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.765242Z","title":"Clara passed the exams, therefore she studied hard","venue":null,"work_id":"e2cdc844-a523-4b12-9512-aedb8181a67a","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.232201Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:76e4ebbd307e0a1f20978dfdcd13aa5df25c71b49ba60cb76f30cdd8d8e87cf3","observation_id":"01fc384f-0306-4105-b761-39bc7b6b7202","resolution":{"observed_at":"2026-08-15T16:16:49.768441Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.756658Z","title":"Riccardo does not play ten- nis, therefore he did not play football (Correct Answer: 3) A.3","venue":null,"work_id":"e49e2ad9-359b-413a-8158-6ef3c7b262df","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.235146Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:5b631f6326795f686dcb68a86462c293572759d033973e89d4355d707e8448cc","observation_id":"b955e2fb-3393-4b54-9464-e3e0decb00da","resolution":{"observed_at":"2026-08-15T16:16:49.760046Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.748134Z","title":null,"venue":null,"work_id":"d827064f-54bc-4a1e-9a0d-f1eebab55741","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.238359Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:aa70dcb63824d4ce5649753fc60fb3fa2def782372f1d2b2cd98b4f9c370ae16","observation_id":"dc5abd4a-3dff-498b-8d32-e9f70ca1bce1","resolution":{"observed_at":"2026-08-15T16:16:49.751081Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.739087Z","title":null,"venue":null,"work_id":"1168c79c-938c-4353-9770-0191abbea0b6","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.241354Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:437c43d488d6a5dfc559eeae6c272adc32839cebf590cf4856b805b61f43c66f","observation_id":"d5b6dd9d-a3bc-4621-86ae-0004792b549b","resolution":{"observed_at":"2026-08-15T16:16:49.741993Z","resolver_source":"raw_fallback","status":"parse_uncertain"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.730009Z","title":null,"venue":null,"work_id":"25f6ed49-dc88-4d7e-9d16-861c26358ca1","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.244491Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:117c8ecea5d46d521afebb54fc23284074c1931a403b1ba3439c14b2c36138e3","observation_id":"06ee3f3e-d069-4cb8-9fb5-17392069896f","resolution":{"observed_at":"2026-08-15T16:16:49.732772Z","resolver_source":"raw_fallback","status":"parse_uncertain"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.721782Z","title":null,"venue":null,"work_id":"614ff233-0609-4730-ac50-2d491b11731f","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.247704Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:7a05a48afe8dfc3648ab2789421170bae3b8d3190176a1cdca2601e0550dbfdb","observation_id":"e3ea7f55-0f8b-4746-b42e-23c6ee37c60e","resolution":{"observed_at":"2026-08-15T16:16:49.724675Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.712464Z","title":null,"venue":null,"work_id":"9fc5512d-23c3-446a-a3eb-e85b5078915a","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.251112Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:f4cdee41f6f6f09e754df5febfb38014adbe4a6f9ff9e0a5059232789c3b5871","observation_id":"b12fa6f6-4463-4830-a9e2-31ee8344aa56","resolution":{"observed_at":"2026-08-15T16:16:49.715691Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.703713Z","title":null,"venue":null,"work_id":"6c876e22-bba8-4614-8266-8ab02c51508b","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.254326Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:b0c8f73769621b27d951c407b414c13d1c41fc50b05bbebb48e23752c15dddae","observation_id":"4d6b9822-df21-4664-aa4d-345e56eca906","resolution":{"observed_at":"2026-08-15T16:16:49.706488Z","resolver_source":"raw_fallback","status":"parse_uncertain"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.695890Z","title":null,"venue":null,"work_id":"47f954c6-831e-4325-bd9c-481bec8f5f39","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.257527Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:e9ae21c2d3edede79e02f9447d61693039509b379d4435a48c1115d25f209120","observation_id":"52b337fb-7527-463e-9865-aeb06616114f","resolution":{"observed_at":"2026-08-15T16:16:49.698528Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.687435Z","title":null,"venue":null,"work_id":"124cd4c5-4607-4d5b-b13b-cbf073bb61da","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.260719Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:242c4152503e9c282bcf53fe9251d0e644ede86ba02485c2c532ffe6e6f048a2","observation_id":"0e9fc822-787c-4c40-a52d-759ec30c0132","resolution":{"observed_at":"2026-08-15T16:16:49.690350Z","resolver_source":"raw_fallback","status":"parse_uncertain"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.678581Z","title":"Chemistry Example Domanda:A quante moli corrispondono 5 mL (d=1,8 g ·cm−3) di un composto avente una massa molare di 450 g·mol−1? Possibili risposte:","venue":null,"work_id":"7ce0fdb1-002f-4265-a725-bf6583aa631c","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.263970Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:2ba099aac3c2340e3ee6e1fbc35b97bc6c3428911908751db95be1daf7248e5c","observation_id":"1e68f2cf-7f91-4d6b-a60a-f724970bb7b3","resolution":{"observed_at":"2026-08-15T16:16:49.681710Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.670088Z","title":null,"venue":null,"work_id":"80c9e647-efde-4f57-976d-42c2fd0b08d7","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.267210Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:e874dfe03f7238b2c2f7de16b3b54fe082a22e75791d316bb5db94e25b0d1c7d","observation_id":"dfa12a9b-b86c-4518-af17-7203273c4e4b","resolution":{"observed_at":"2026-08-15T16:16:49.672857Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.661063Z","title":null,"venue":null,"work_id":"1f3a8453-a085-4c18-b06c-577fdf441daa","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.271160Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:09ae271f5e7b19d1cb297f172ecb1a641418b2c1e503473384d72f4f78c28ec0","observation_id":"f7dd84eb-8475-456b-97b8-c15416b26e80","resolution":{"observed_at":"2026-08-15T16:16:49.663843Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.652428Z","title":null,"venue":null,"work_id":"d3036b7d-ca57-44c7-b1ca-c1ce492fc2cd","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.274285Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:81b886646e1bfc4ba5b5e55c9bd328cee400a3223642d101415aa39a2c5ef7bc","observation_id":"4465b9ea-ee69-4b40-8315-7483de6bb1ed","resolution":{"observed_at":"2026-08-15T16:16:49.655370Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.642530Z","title":null,"venue":null,"work_id":"a9beb3f7-a213-4b9a-bb77-e6759b61e80f","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.278030Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:ce6f9668de8c5a1a7a0c7db33ad67d9c7461414ed4db5ba3108c34482ac7f3d2","observation_id":"4aa63aa9-1018-4f60-8830-53869bab582b","resolution":{"observed_at":"2026-08-15T16:16:49.645513Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.632870Z","title":null,"venue":null,"work_id":"34585f01-e0df-4057-8267-ffcb0053cbd0","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.281242Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:fd198fcf9fbdde88dfcae60d89dca7738ad95dba72ebec2871c029b3686265d8","observation_id":"d527f1af-176b-460f-a817-8d8c7490ab44","resolution":{"observed_at":"2026-08-15T16:16:49.636019Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.622800Z","title":"Mathematics Example Domanda:Dati tre segmenti AA’, BB’ e CC’ tali che: AA’ = 2 cm, BB’ = 1,5 * AA’, CC’ = 2,0 * BB’","venue":null,"work_id":"5a207d35-2db6-487e-99ec-5e9b6a65ab8f","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.284391Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:1d7c8979890af660822263df4854cc81dba1280fda33041d7a42531e2a5bd002","observation_id":"121ac5b2-ba64-47a5-84bc-0df35dfabf48","resolution":{"observed_at":"2026-08-15T16:16:49.626694Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.613323Z","title":null,"venue":null,"work_id":"80cc3754-2b58-45bc-9a6a-ed1834fe2891","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.287982Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:aef599c4320176defd9e1b7e555743ca2c51a4c60d5b9d49764458ac062a9eae","observation_id":"beb19267-748c-4cab-99ec-03486af1a569","resolution":{"observed_at":"2026-08-15T16:16:49.616832Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.603987Z","title":null,"venue":null,"work_id":"17c7496e-1343-4fb0-945a-66f6937dd0b4","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.290935Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:78fd649c0bcfbbe222c59101b2a19dd5ab91cb6f3ba01ab09db2211f7b09fdf6","observation_id":"98692b07-e034-4301-b79a-aca303a168cf","resolution":{"observed_at":"2026-08-15T16:16:49.607039Z","resolver_source":"raw_fallback","status":"parse_uncertain"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.593987Z","title":null,"venue":null,"work_id":"897ecf29-ea6e-40ce-b453-76f09bf46a21","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.293642Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:2c84f69e4b15b6b786e109aad61025fd0257ef09f5357ea41470a5f2133c862b","observation_id":"4d423c89-8307-4191-b0c9-6ecccd559b36","resolution":{"observed_at":"2026-08-15T16:16:49.597102Z","resolver_source":"raw_fallback","status":"parse_uncertain"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.583738Z","title":null,"venue":null,"work_id":"114e03f2-6f92-4e04-924d-03bcb0deeb2a","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.296835Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:5749793410bad4227618c7960da73abb243441b31673c94169c13ae1d2d9760f","observation_id":"0e97a9d2-5a33-4ca5-8f7f-f38a82323d18","resolution":{"observed_at":"2026-08-15T16:16:49.586934Z","resolver_source":"raw_fallback","status":"parse_uncertain"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.572752Z","title":"Which triangle is possible to construct with these sides? Possible answers:","venue":null,"work_id":"deae0e99-4997-475a-837e-bbaf854e1342","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.299876Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:cebc1a0389e57bf59680ecb51dd7eaa4903bd58d1782f933505612bd9f99bf3a","observation_id":"a02d19f7-f5bd-43fb-8159-821f7c13b79a","resolution":{"observed_at":"2026-08-15T16:16:49.576247Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.559950Z","title":null,"venue":null,"work_id":"6022395b-e941-4fd3-b9f2-31cf482e53e4","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.302833Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:5795f77b5b8dceb1801f9db60240679c8e3211df75cfc73e7d9c2c00a6a2c882","observation_id":"2acda169-967c-40c5-b247-e08c418936e8","resolution":{"observed_at":"2026-08-15T16:16:49.563265Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.549525Z","title":null,"venue":null,"work_id":"8b446f23-c0e6-469f-9e30-5919796ea5df","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.306274Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:9b3589192b0612adeff2ad14713b8b6ae21cd54d465b45f513f61669f3e3a698","observation_id":"61a9a47e-e0f4-481b-8d34-bdc0bb827362","resolution":{"observed_at":"2026-08-15T16:16:49.552433Z","resolver_source":"raw_fallback","status":"parse_uncertain"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.539757Z","title":null,"venue":null,"work_id":"9350c729-b3ef-44d1-a555-884e0f92986e","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.309383Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:2f359a4dd679fc284e6fcf76f4841a42af61b5586958e3af8e68f6ad577e3477","observation_id":"bd18b8fd-9f9f-4327-beaa-2e163ca0c557","resolution":{"observed_at":"2026-08-15T16:16:49.542972Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:16:49.526410Z","title":"Per-Subject Model Performance Table 4 Per-subject accuracy (%) on MedBench-IT for Standard (Std.) and Reasoning (Reas.) prompts","venue":null,"work_id":"a761d65c-4038-49fe-bb87-d7d73f5a84c2","year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.312244Z"},"links":{"citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:a69b864bce091572d63f5e6af1b44beb6c6d7116b4b485aef7d4eccbd039fdf5","observation_id":"b8559297-e408-42e4-be3c-c5cf45afb821","resolution":{"observed_at":"2026-08-15T16:16:49.531635Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2009.03300","last_updated":"2021-01-12T18:57:11Z","snapshot_observed_at":"2026-08-13T20:44:28.824685Z","submitted_at":"2020-09-07T17:59:25Z","title":"Measuring Massive Multitask Language Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.03300","snapshot_observed_at":"2026-08-15T16:16:49.093791Z","title":"doi:10.48550/arXiv.2009.03300, arXiv:2009.03300 [cs]","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.093791Z"},"links":{"cited_paper":"/paper/2009.03300","citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:c56013715c71c1909a67e11df2ef7a574c6010ac7f39d0eceef17f7ba8e1fe74","observation_id":"612f47fa-c3d0-476a-80cc-d7b96b97530e","resolution":{"observed_at":"2026-08-15T16:16:49.093791Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2201.11903","last_updated":"2023-01-10T23:07:57Z","snapshot_observed_at":"2026-08-13T07:04:41.220509Z","submitted_at":"2022-01-28T02:33:07Z","title":"Chain-of-Thought Prompting Elicits Reasoning in Large Language Models","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2201.11903","snapshot_observed_at":"2026-08-15T16:16:49.147542Z","title":"doi:10.48550/arXiv.2201.11903, arXiv:2201.11903 [cs]","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.147542Z"},"links":{"cited_paper":"/paper/2201.11903","citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:e93dffd4c10c7d351001db1a34420c8e05e4bf8e3b321ac047d2375d25ba183d","observation_id":"36afa985-006f-4174-a06b-ab01ab22e8ac","resolution":{"observed_at":"2026-08-15T16:16:49.147542Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-15T16:16:49.170935Z","title":"doi:10.48550/arXiv.2407.21783, arXiv:2407.21783 [cs]","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.170935Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:6fc1fe7515f05aa98776ebd7a8c87cc61422be2b12e2f0ea4a5e23cc2d4da12a","observation_id":"b483a8fa-bc0c-4e15-bb33-0fc1cc6dc9c4","resolution":{"observed_at":"2026-08-15T16:16:49.170935Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-08-17T18:50:07.059564Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-15T16:16:49.160095Z","title":"arXiv:2412.15115","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-15T16:16:49.160095Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2509.07135"},"observation_digest":"sha256:fffbb61d1b4c3215637e8a462ac024dc86c97d85c8e47475bb3c1b36411caeda","observation_id":"eb6dd17c-2406-4603-9855-6aa9ff542f6c","resolution":{"observed_at":"2026-08-15T16:16:49.160095Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2509.07135","last_updated":"2025-09-08T18:39:35Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-18T14:25:33.098668Z","submitted_at":"2025-09-08T18:39:35Z","title":"MedBench-IT: A Comprehensive Benchmark for Evaluating Large Language Models on Italian Medical Entrance Examinations"},"reference_resolution":{"displayed":71,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":8,"unresolved":32,"verified_exact":0,"verified_fuzzy":31},"total_outbound_references":71},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"thesis":"As of 23 August 2026, this Paper Citation Record lists 71 of 71 outbound references and 0 inbound Pith citation observations for arXiv:2509.07135."}