{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:I5GR3MWXG6ITTV3FTOMNFBDHXD","short_pith_number":"pith:I5GR3MWX","schema_version":"1.0","canonical_sha256":"474d1db2d7379139d7659b98d28467b8fb6a11707b153ac1c357605211bc7698","source":{"kind":"arxiv","id":"2402.09654","version":2},"attestation_state":"computed","paper":{"title":"GPT-4's assessment of its performance in a USMLE-based case study","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.HC","cs.MA","stat.ML"],"primary_cat":"cs.AI","authors_text":"Aniket Kumar Singh, Bishal Lamichhane, Chandra Dhakal, Suman Devkota, Suprinsa Paudyal, Uttam Dhakal, Yogesh Sapkota","submitted_at":"2024-02-15T01:38:50Z","abstract_excerpt":"This study investigates GPT-4's assessment of its performance in healthcare applications. A simple prompting technique was used to prompt the LLM with questions taken from the United States Medical Licensing Examination (USMLE) questionnaire and it was tasked to evaluate its confidence score before posing the question and after asking the question. The questionnaire was categorized into two groups-questions with feedback (WF) and questions with no feedback(NF) post-question. The model was asked to provide absolute and relative confidence scores before and after each question. The experimental "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.09654","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-02-15T01:38:50Z","cross_cats_sorted":["cs.CL","cs.HC","cs.MA","stat.ML"],"title_canon_sha256":"8e33c24cbd974993e2776d6ef63b55c456f5f9537b3cdc0444fd9e92233df2ae","abstract_canon_sha256":"b5334dbc4e64deda0c5a7ad951d508d61c64b393b411dffd095c440c6268225c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:01:01.584401Z","signature_b64":"01pcYE0jSvTjxKxawcInaHwyJ+wwfaVEhTOktllZaY4dVeQNWzoDXz71g8L/1K3DlpqUGS+His2NEhl5tlHRAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"474d1db2d7379139d7659b98d28467b8fb6a11707b153ac1c357605211bc7698","last_reissued_at":"2026-07-05T08:01:01.583898Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:01:01.583898Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GPT-4's assessment of its performance in a USMLE-based case study","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.HC","cs.MA","stat.ML"],"primary_cat":"cs.AI","authors_text":"Aniket Kumar Singh, Bishal Lamichhane, Chandra Dhakal, Suman Devkota, Suprinsa Paudyal, Uttam Dhakal, Yogesh Sapkota","submitted_at":"2024-02-15T01:38:50Z","abstract_excerpt":"This study investigates GPT-4's assessment of its performance in healthcare applications. A simple prompting technique was used to prompt the LLM with questions taken from the United States Medical Licensing Examination (USMLE) questionnaire and it was tasked to evaluate its confidence score before posing the question and after asking the question. The questionnaire was categorized into two groups-questions with feedback (WF) and questions with no feedback(NF) post-question. The model was asked to provide absolute and relative confidence scores before and after each question. The experimental "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.09654","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.09654/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.09654","created_at":"2026-07-05T08:01:01.583955+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.09654v2","created_at":"2026-07-05T08:01:01.583955+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.09654","created_at":"2026-07-05T08:01:01.583955+00:00"},{"alias_kind":"pith_short_12","alias_value":"I5GR3MWXG6IT","created_at":"2026-07-05T08:01:01.583955+00:00"},{"alias_kind":"pith_short_16","alias_value":"I5GR3MWXG6ITTV3F","created_at":"2026-07-05T08:01:01.583955+00:00"},{"alias_kind":"pith_short_8","alias_value":"I5GR3MWX","created_at":"2026-07-05T08:01:01.583955+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/I5GR3MWXG6ITTV3FTOMNFBDHXD","json":"https://pith.science/pith/I5GR3MWXG6ITTV3FTOMNFBDHXD.json","graph_json":"https://pith.science/api/pith-number/I5GR3MWXG6ITTV3FTOMNFBDHXD/graph.json","events_json":"https://pith.science/api/pith-number/I5GR3MWXG6ITTV3FTOMNFBDHXD/events.json","paper":"https://pith.science/paper/I5GR3MWX"},"agent_actions":{"view_html":"https://pith.science/pith/I5GR3MWXG6ITTV3FTOMNFBDHXD","download_json":"https://pith.science/pith/I5GR3MWXG6ITTV3FTOMNFBDHXD.json","view_paper":"https://pith.science/paper/I5GR3MWX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.09654&json=true","fetch_graph":"https://pith.science/api/pith-number/I5GR3MWXG6ITTV3FTOMNFBDHXD/graph.json","fetch_events":"https://pith.science/api/pith-number/I5GR3MWXG6ITTV3FTOMNFBDHXD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/I5GR3MWXG6ITTV3FTOMNFBDHXD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/I5GR3MWXG6ITTV3FTOMNFBDHXD/action/storage_attestation","attest_author":"https://pith.science/pith/I5GR3MWXG6ITTV3FTOMNFBDHXD/action/author_attestation","sign_citation":"https://pith.science/pith/I5GR3MWXG6ITTV3FTOMNFBDHXD/action/citation_signature","submit_replication":"https://pith.science/pith/I5GR3MWXG6ITTV3FTOMNFBDHXD/action/replication_record"}},"created_at":"2026-07-05T08:01:01.583955+00:00","updated_at":"2026-07-05T08:01:01.583955+00:00"}