{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:L2SOHRHTDTYCDVU7QVYWJTQFNN","short_pith_number":"pith:L2SOHRHT","schema_version":"1.0","canonical_sha256":"5ea4e3c4f31cf021d69f857164ce056b61e8dc6c55022f843adcbbd595240629","source":{"kind":"arxiv","id":"2505.11887","version":1},"attestation_state":"computed","paper":{"title":"AutoMedEval: Harnessing Language Models for Automatic Medical Capability Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Gerard de Melo, Liang He, Linlin Wang, Xiaoling Wang, Xiechi Zhang, Yanfeng Wang, Ya Zhang, Zetian Ouyang, Zhu Cao","submitted_at":"2025-05-17T07:44:54Z","abstract_excerpt":"With the proliferation of large language models (LLMs) in the medical domain, there is increasing demand for improved evaluation techniques to assess their capabilities. However, traditional metrics like F1 and ROUGE, which rely on token overlaps to measure quality, significantly overlook the importance of medical terminology. While human evaluation tends to be more reliable, it can be very costly and may as well suffer from inaccuracies due to limits in human expertise and motivation. Although there are some evaluation methods based on LLMs, their usability in the medical field is limited due"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.11887","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-05-17T07:44:54Z","cross_cats_sorted":[],"title_canon_sha256":"872239e82ce0f09947c73a1e08d810faa59a2bedeef190777325590965f9a94f","abstract_canon_sha256":"a6efbe8f8f3af9f1f7361cd9eb07c8b9e514059679cc79b42389cf58d1e2d25d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:04:53.883188Z","signature_b64":"oW5qXqNHeexuPfGTRrnIfz7egDG0INCqI+rEpP54MQx7cVmBd5LafyWPAj7HLRT00I2hpwjc+h5tObPG4LXTCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5ea4e3c4f31cf021d69f857164ce056b61e8dc6c55022f843adcbbd595240629","last_reissued_at":"2026-07-05T11:04:53.882662Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:04:53.882662Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AutoMedEval: Harnessing Language Models for Automatic Medical Capability Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Gerard de Melo, Liang He, Linlin Wang, Xiaoling Wang, Xiechi Zhang, Yanfeng Wang, Ya Zhang, Zetian Ouyang, Zhu Cao","submitted_at":"2025-05-17T07:44:54Z","abstract_excerpt":"With the proliferation of large language models (LLMs) in the medical domain, there is increasing demand for improved evaluation techniques to assess their capabilities. However, traditional metrics like F1 and ROUGE, which rely on token overlaps to measure quality, significantly overlook the importance of medical terminology. While human evaluation tends to be more reliable, it can be very costly and may as well suffer from inaccuracies due to limits in human expertise and motivation. Although there are some evaluation methods based on LLMs, their usability in the medical field is limited due"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.11887","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.11887/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.11887","created_at":"2026-07-05T11:04:53.882723+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.11887v1","created_at":"2026-07-05T11:04:53.882723+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.11887","created_at":"2026-07-05T11:04:53.882723+00:00"},{"alias_kind":"pith_short_12","alias_value":"L2SOHRHTDTYC","created_at":"2026-07-05T11:04:53.882723+00:00"},{"alias_kind":"pith_short_16","alias_value":"L2SOHRHTDTYCDVU7","created_at":"2026-07-05T11:04:53.882723+00:00"},{"alias_kind":"pith_short_8","alias_value":"L2SOHRHT","created_at":"2026-07-05T11:04:53.882723+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/L2SOHRHTDTYCDVU7QVYWJTQFNN","json":"https://pith.science/pith/L2SOHRHTDTYCDVU7QVYWJTQFNN.json","graph_json":"https://pith.science/api/pith-number/L2SOHRHTDTYCDVU7QVYWJTQFNN/graph.json","events_json":"https://pith.science/api/pith-number/L2SOHRHTDTYCDVU7QVYWJTQFNN/events.json","paper":"https://pith.science/paper/L2SOHRHT"},"agent_actions":{"view_html":"https://pith.science/pith/L2SOHRHTDTYCDVU7QVYWJTQFNN","download_json":"https://pith.science/pith/L2SOHRHTDTYCDVU7QVYWJTQFNN.json","view_paper":"https://pith.science/paper/L2SOHRHT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.11887&json=true","fetch_graph":"https://pith.science/api/pith-number/L2SOHRHTDTYCDVU7QVYWJTQFNN/graph.json","fetch_events":"https://pith.science/api/pith-number/L2SOHRHTDTYCDVU7QVYWJTQFNN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/L2SOHRHTDTYCDVU7QVYWJTQFNN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/L2SOHRHTDTYCDVU7QVYWJTQFNN/action/storage_attestation","attest_author":"https://pith.science/pith/L2SOHRHTDTYCDVU7QVYWJTQFNN/action/author_attestation","sign_citation":"https://pith.science/pith/L2SOHRHTDTYCDVU7QVYWJTQFNN/action/citation_signature","submit_replication":"https://pith.science/pith/L2SOHRHTDTYCDVU7QVYWJTQFNN/action/replication_record"}},"created_at":"2026-07-05T11:04:53.882723+00:00","updated_at":"2026-07-05T11:04:53.882723+00:00"}