{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:XOJ3ESFGXK2RPL53IWZ4HYTDKW","short_pith_number":"pith:XOJ3ESFG","schema_version":"1.0","canonical_sha256":"bb93b248a6bab517afbb45b3c3e26355a8a4026cf7b8afaeb4fa7b4605fbef8b","source":{"kind":"arxiv","id":"2506.21578","version":1},"attestation_state":"computed","paper":{"title":"HealthQA-BR: A System-Wide Benchmark Reveals Critical Knowledge Gaps in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Andrew Maranh\\~ao Ventura D'addario","submitted_at":"2025-06-16T07:40:25Z","abstract_excerpt":"The evaluation of Large Language Models (LLMs) in healthcare has been dominated by physician-centric, English-language benchmarks, creating a dangerous illusion of competence that ignores the interprofessional nature of patient care. To provide a more holistic and realistic assessment, we introduce HealthQA-BR, the first large-scale, system-wide benchmark for Portuguese-speaking healthcare. Comprising 5,632 questions from Brazil's national licensing and residency exams, it uniquely assesses knowledge not only in medicine and its specialties but also in nursing, dentistry, psychology, social wo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.21578","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-16T07:40:25Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"17f133c3cb2985530a37816f17ad16f1f064e4b906fdc0bc72fc8923533d9a63","abstract_canon_sha256":"a913c6d91ef92a115e9e66ab057b114e89a71a53e42df63dcbac5514227da147"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:27:56.677446Z","signature_b64":"Suv26Mdctqwm1mdp7U2jUORQckf0yheAyNTAINF4WCCu8wdvc3Vc4yU/BYBhVJZ3HRthH/SUY88V+nW3cNztDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bb93b248a6bab517afbb45b3c3e26355a8a4026cf7b8afaeb4fa7b4605fbef8b","last_reissued_at":"2026-07-05T11:27:56.676978Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:27:56.676978Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HealthQA-BR: A System-Wide Benchmark Reveals Critical Knowledge Gaps in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Andrew Maranh\\~ao Ventura D'addario","submitted_at":"2025-06-16T07:40:25Z","abstract_excerpt":"The evaluation of Large Language Models (LLMs) in healthcare has been dominated by physician-centric, English-language benchmarks, creating a dangerous illusion of competence that ignores the interprofessional nature of patient care. To provide a more holistic and realistic assessment, we introduce HealthQA-BR, the first large-scale, system-wide benchmark for Portuguese-speaking healthcare. Comprising 5,632 questions from Brazil's national licensing and residency exams, it uniquely assesses knowledge not only in medicine and its specialties but also in nursing, dentistry, psychology, social wo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.21578","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.21578/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.21578","created_at":"2026-07-05T11:27:56.677039+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.21578v1","created_at":"2026-07-05T11:27:56.677039+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.21578","created_at":"2026-07-05T11:27:56.677039+00:00"},{"alias_kind":"pith_short_12","alias_value":"XOJ3ESFGXK2R","created_at":"2026-07-05T11:27:56.677039+00:00"},{"alias_kind":"pith_short_16","alias_value":"XOJ3ESFGXK2RPL53","created_at":"2026-07-05T11:27:56.677039+00:00"},{"alias_kind":"pith_short_8","alias_value":"XOJ3ESFG","created_at":"2026-07-05T11:27:56.677039+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06641","citing_title":"Healthier LLMs: Retrieval-Augmented Generation for Public Health Question Answering","ref_index":17,"is_internal_anchor":true},{"citing_arxiv_id":"2605.01077","citing_title":"Teaching LLMs Brazilian Healthcare: Injecting Knowledge from Official Clinical Guidelines","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XOJ3ESFGXK2RPL53IWZ4HYTDKW","json":"https://pith.science/pith/XOJ3ESFGXK2RPL53IWZ4HYTDKW.json","graph_json":"https://pith.science/api/pith-number/XOJ3ESFGXK2RPL53IWZ4HYTDKW/graph.json","events_json":"https://pith.science/api/pith-number/XOJ3ESFGXK2RPL53IWZ4HYTDKW/events.json","paper":"https://pith.science/paper/XOJ3ESFG"},"agent_actions":{"view_html":"https://pith.science/pith/XOJ3ESFGXK2RPL53IWZ4HYTDKW","download_json":"https://pith.science/pith/XOJ3ESFGXK2RPL53IWZ4HYTDKW.json","view_paper":"https://pith.science/paper/XOJ3ESFG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.21578&json=true","fetch_graph":"https://pith.science/api/pith-number/XOJ3ESFGXK2RPL53IWZ4HYTDKW/graph.json","fetch_events":"https://pith.science/api/pith-number/XOJ3ESFGXK2RPL53IWZ4HYTDKW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XOJ3ESFGXK2RPL53IWZ4HYTDKW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XOJ3ESFGXK2RPL53IWZ4HYTDKW/action/storage_attestation","attest_author":"https://pith.science/pith/XOJ3ESFGXK2RPL53IWZ4HYTDKW/action/author_attestation","sign_citation":"https://pith.science/pith/XOJ3ESFGXK2RPL53IWZ4HYTDKW/action/citation_signature","submit_replication":"https://pith.science/pith/XOJ3ESFGXK2RPL53IWZ4HYTDKW/action/replication_record"}},"created_at":"2026-07-05T11:27:56.677039+00:00","updated_at":"2026-07-05T11:27:56.677039+00:00"}