{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:2PQ5EM6DFKUIQVCHXPZCWALL4I","short_pith_number":"pith:2PQ5EM6D","schema_version":"1.0","canonical_sha256":"d3e1d233c32aa8885447bbf22b016be20e8a3fe6e3af396ab13a8bcccc7e5dd9","source":{"kind":"arxiv","id":"2403.08495","version":4},"attestation_state":"computed","paper":{"title":"Automatic Interactive Evaluation for Large Language Models with State Aware Patient Simulator","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hongcheng Liu, Yanfeng Wang, Yuhao Wang, Yusheng Liao, Yutong Meng, Yu Wang","submitted_at":"2024-03-13T13:04:58Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated remarkable proficiency in human interactions, yet their application within the medical field remains insufficiently explored. Previous works mainly focus on the performance of medical knowledge with examinations, which is far from the realistic scenarios, falling short in assessing the abilities of LLMs on clinical tasks. In the quest to enhance the application of Large Language Models (LLMs) in healthcare, this paper introduces the Automated Interactive Evaluation (AIE) framework and the State-Aware Patient Simulator (SAPS), targeting the gap bet"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.08495","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-03-13T13:04:58Z","cross_cats_sorted":[],"title_canon_sha256":"9555276c864ac419ac66c4b3f77851071f5500ffbde7834b219b6701f7a9ca07","abstract_canon_sha256":"7a5a3287901d640c460429a91abc9f18fe351648905ee9f245525a6403ab4093"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:46:22.792232Z","signature_b64":"j0Cg2xRYtzsomkFdVlN6C/7IvhuXzHvYSIms+zOseaVAU5tQ8t27PfqUiupXwrI8te8E46xtUxtxdFz9/7KRCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d3e1d233c32aa8885447bbf22b016be20e8a3fe6e3af396ab13a8bcccc7e5dd9","last_reissued_at":"2026-07-05T08:46:22.791803Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:46:22.791803Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Automatic Interactive Evaluation for Large Language Models with State Aware Patient Simulator","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hongcheng Liu, Yanfeng Wang, Yuhao Wang, Yusheng Liao, Yutong Meng, Yu Wang","submitted_at":"2024-03-13T13:04:58Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated remarkable proficiency in human interactions, yet their application within the medical field remains insufficiently explored. Previous works mainly focus on the performance of medical knowledge with examinations, which is far from the realistic scenarios, falling short in assessing the abilities of LLMs on clinical tasks. In the quest to enhance the application of Large Language Models (LLMs) in healthcare, this paper introduces the Automated Interactive Evaluation (AIE) framework and the State-Aware Patient Simulator (SAPS), targeting the gap bet"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.08495","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.08495/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.08495","created_at":"2026-07-05T08:46:22.791861+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.08495v4","created_at":"2026-07-05T08:46:22.791861+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.08495","created_at":"2026-07-05T08:46:22.791861+00:00"},{"alias_kind":"pith_short_12","alias_value":"2PQ5EM6DFKUI","created_at":"2026-07-05T08:46:22.791861+00:00"},{"alias_kind":"pith_short_16","alias_value":"2PQ5EM6DFKUIQVCH","created_at":"2026-07-05T08:46:22.791861+00:00"},{"alias_kind":"pith_short_8","alias_value":"2PQ5EM6D","created_at":"2026-07-05T08:46:22.791861+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17441","citing_title":"Patients With Personality: Realistic Patient Simulation through Controlled Diversity and Selective Disclosure","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22047","citing_title":"Active Evidence-Seeking and Diagnostic Reasoning in Large Language Models for Clinical Decision Support","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2405.07960","citing_title":"AgentClinic: a multimodal agent benchmark to evaluate AI in simulated clinical environments","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20022","citing_title":"MoBayes: A Modular Bayesian Framework for Separating Reasoning from Language in Conversational Clinical Decision Support","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2601.10970","citing_title":"Simulating Couple Conflict: Designing A Multi-Agent System for Therapy Training and Practice","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20022","citing_title":"MoBayes: A Modular Bayesian Framework for Separating Reasoning from Language in Conversational Clinical Decision Support","ref_index":32,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2PQ5EM6DFKUIQVCHXPZCWALL4I","json":"https://pith.science/pith/2PQ5EM6DFKUIQVCHXPZCWALL4I.json","graph_json":"https://pith.science/api/pith-number/2PQ5EM6DFKUIQVCHXPZCWALL4I/graph.json","events_json":"https://pith.science/api/pith-number/2PQ5EM6DFKUIQVCHXPZCWALL4I/events.json","paper":"https://pith.science/paper/2PQ5EM6D"},"agent_actions":{"view_html":"https://pith.science/pith/2PQ5EM6DFKUIQVCHXPZCWALL4I","download_json":"https://pith.science/pith/2PQ5EM6DFKUIQVCHXPZCWALL4I.json","view_paper":"https://pith.science/paper/2PQ5EM6D","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.08495&json=true","fetch_graph":"https://pith.science/api/pith-number/2PQ5EM6DFKUIQVCHXPZCWALL4I/graph.json","fetch_events":"https://pith.science/api/pith-number/2PQ5EM6DFKUIQVCHXPZCWALL4I/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2PQ5EM6DFKUIQVCHXPZCWALL4I/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2PQ5EM6DFKUIQVCHXPZCWALL4I/action/storage_attestation","attest_author":"https://pith.science/pith/2PQ5EM6DFKUIQVCHXPZCWALL4I/action/author_attestation","sign_citation":"https://pith.science/pith/2PQ5EM6DFKUIQVCHXPZCWALL4I/action/citation_signature","submit_replication":"https://pith.science/pith/2PQ5EM6DFKUIQVCHXPZCWALL4I/action/replication_record"}},"created_at":"2026-07-05T08:46:22.791861+00:00","updated_at":"2026-07-05T08:46:22.791861+00:00"}