{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LEITPM5UVC42OOUDYVLOCLCQYL","short_pith_number":"pith:LEITPM5U","schema_version":"1.0","canonical_sha256":"591137b3b4a8b9a73a83c556e12c50c2c794ed7b943b640c651f791eec491311","source":{"kind":"arxiv","id":"2501.06741","version":1},"attestation_state":"computed","paper":{"title":"Hierarchical Divide-and-Conquer for Fine-Grained Alignment in LLM-Based Medical Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Gerard de Melo, Linlin Wang, Shunfan Zheng, Xiaoling Wang, Xiechi Zhang","submitted_at":"2025-01-12T07:30:49Z","abstract_excerpt":"In the rapidly evolving landscape of large language models (LLMs) for medical applications, ensuring the reliability and accuracy of these models in clinical settings is paramount. Existing benchmarks often focus on fixed-format tasks like multiple-choice QA, which fail to capture the complexity of real-world clinical diagnostics. Moreover, traditional evaluation metrics and LLM-based evaluators struggle with misalignment, often providing oversimplified assessments that do not adequately reflect human judgment. To address these challenges, we introduce HDCEval, a Hierarchical Divide-and-Conque"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.06741","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-01-12T07:30:49Z","cross_cats_sorted":[],"title_canon_sha256":"0503884f86201d6568580737b109720880a0c70a72007d705b2f540f3439c959","abstract_canon_sha256":"e3de4018b91b60efe45289c6a01d5dec8d853014d7bfcc8e44bb56bf0758d10f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:00:09.388860Z","signature_b64":"5ckR7hBf34kIKUAyxBohZ+u8Rh5dVpTLh2NtQKlHdBVHyvAX7YAA0jyhDKGoVreU28+BF+SBqo86ivejOEl1DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"591137b3b4a8b9a73a83c556e12c50c2c794ed7b943b640c651f791eec491311","last_reissued_at":"2026-07-05T10:00:09.388383Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:00:09.388383Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Hierarchical Divide-and-Conquer for Fine-Grained Alignment in LLM-Based Medical Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Gerard de Melo, Linlin Wang, Shunfan Zheng, Xiaoling Wang, Xiechi Zhang","submitted_at":"2025-01-12T07:30:49Z","abstract_excerpt":"In the rapidly evolving landscape of large language models (LLMs) for medical applications, ensuring the reliability and accuracy of these models in clinical settings is paramount. Existing benchmarks often focus on fixed-format tasks like multiple-choice QA, which fail to capture the complexity of real-world clinical diagnostics. Moreover, traditional evaluation metrics and LLM-based evaluators struggle with misalignment, often providing oversimplified assessments that do not adequately reflect human judgment. To address these challenges, we introduce HDCEval, a Hierarchical Divide-and-Conque"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.06741","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.06741/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.06741","created_at":"2026-07-05T10:00:09.388442+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.06741v1","created_at":"2026-07-05T10:00:09.388442+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.06741","created_at":"2026-07-05T10:00:09.388442+00:00"},{"alias_kind":"pith_short_12","alias_value":"LEITPM5UVC42","created_at":"2026-07-05T10:00:09.388442+00:00"},{"alias_kind":"pith_short_16","alias_value":"LEITPM5UVC42OOUD","created_at":"2026-07-05T10:00:09.388442+00:00"},{"alias_kind":"pith_short_8","alias_value":"LEITPM5U","created_at":"2026-07-05T10:00:09.388442+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.20824","citing_title":"MedSentry: Understanding and Mitigating Safety Risks in Medical LLM Multi-Agent Systems","ref_index":55,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LEITPM5UVC42OOUDYVLOCLCQYL","json":"https://pith.science/pith/LEITPM5UVC42OOUDYVLOCLCQYL.json","graph_json":"https://pith.science/api/pith-number/LEITPM5UVC42OOUDYVLOCLCQYL/graph.json","events_json":"https://pith.science/api/pith-number/LEITPM5UVC42OOUDYVLOCLCQYL/events.json","paper":"https://pith.science/paper/LEITPM5U"},"agent_actions":{"view_html":"https://pith.science/pith/LEITPM5UVC42OOUDYVLOCLCQYL","download_json":"https://pith.science/pith/LEITPM5UVC42OOUDYVLOCLCQYL.json","view_paper":"https://pith.science/paper/LEITPM5U","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.06741&json=true","fetch_graph":"https://pith.science/api/pith-number/LEITPM5UVC42OOUDYVLOCLCQYL/graph.json","fetch_events":"https://pith.science/api/pith-number/LEITPM5UVC42OOUDYVLOCLCQYL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LEITPM5UVC42OOUDYVLOCLCQYL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LEITPM5UVC42OOUDYVLOCLCQYL/action/storage_attestation","attest_author":"https://pith.science/pith/LEITPM5UVC42OOUDYVLOCLCQYL/action/author_attestation","sign_citation":"https://pith.science/pith/LEITPM5UVC42OOUDYVLOCLCQYL/action/citation_signature","submit_replication":"https://pith.science/pith/LEITPM5UVC42OOUDYVLOCLCQYL/action/replication_record"}},"created_at":"2026-07-05T10:00:09.388442+00:00","updated_at":"2026-07-05T10:00:09.388442+00:00"}