{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:EFNXFH6QZS64RMYW5C53RHPED2","short_pith_number":"pith:EFNXFH6Q","schema_version":"1.0","canonical_sha256":"215b729fd0ccbdc8b316e8bbb89de41eb355f85a45635d2ffcf892c3fd696c8d","source":{"kind":"arxiv","id":"2403.20180","version":1},"attestation_state":"computed","paper":{"title":"Measuring Taiwanese Mandarin Language Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Po-Heng Chen, Sijia Cheng, Wei-Lin Chen, Yen-Ting Lin, Yun-Nung Chen","submitted_at":"2024-03-29T13:56:21Z","abstract_excerpt":"The evaluation of large language models (LLMs) has drawn substantial attention in the field recently. This work focuses on evaluating LLMs in a Chinese context, specifically, for Traditional Chinese which has been largely underrepresented in existing benchmarks. We present TMLU, a holistic evaluation suit tailored for assessing the advanced knowledge and reasoning capability in LLMs, under the context of Taiwanese Mandarin. TMLU consists of an array of 37 subjects across social science, STEM, humanities, Taiwan-specific content, and others, ranging from middle school to professional levels. In"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.20180","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-03-29T13:56:21Z","cross_cats_sorted":[],"title_canon_sha256":"dd3a949a6e67e5de628fa8aea9b9764b7e8c6b81cf48495a6a9c950fee77a23c","abstract_canon_sha256":"a3afa19d69ea489fdc60dead276abf71cdf63dbbce9430219c51fb3f3cd2abd6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:02:15.325982Z","signature_b64":"/fNyGVn1AVCYgUGegAvGf2WI18X0jal7i6rBE9d+9xyedXurHC4gEVTeB/stC96E7faxnJsCpQLGmXPswYa6Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"215b729fd0ccbdc8b316e8bbb89de41eb355f85a45635d2ffcf892c3fd696c8d","last_reissued_at":"2026-07-05T08:02:15.325403Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:02:15.325403Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Measuring Taiwanese Mandarin Language Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Po-Heng Chen, Sijia Cheng, Wei-Lin Chen, Yen-Ting Lin, Yun-Nung Chen","submitted_at":"2024-03-29T13:56:21Z","abstract_excerpt":"The evaluation of large language models (LLMs) has drawn substantial attention in the field recently. This work focuses on evaluating LLMs in a Chinese context, specifically, for Traditional Chinese which has been largely underrepresented in existing benchmarks. We present TMLU, a holistic evaluation suit tailored for assessing the advanced knowledge and reasoning capability in LLMs, under the context of Taiwanese Mandarin. TMLU consists of an array of 37 subjects across social science, STEM, humanities, Taiwan-specific content, and others, ranging from middle school to professional levels. In"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.20180","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.20180/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.20180","created_at":"2026-07-05T08:02:15.325469+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.20180v1","created_at":"2026-07-05T08:02:15.325469+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.20180","created_at":"2026-07-05T08:02:15.325469+00:00"},{"alias_kind":"pith_short_12","alias_value":"EFNXFH6QZS64","created_at":"2026-07-05T08:02:15.325469+00:00"},{"alias_kind":"pith_short_16","alias_value":"EFNXFH6QZS64RMYW","created_at":"2026-07-05T08:02:15.325469+00:00"},{"alias_kind":"pith_short_8","alias_value":"EFNXFH6Q","created_at":"2026-07-05T08:02:15.325469+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06054","citing_title":"BlueMagpie-TTS: A Token-Efficient Tokenizer, Language Model, and TTS for Taiwanese-Accent Code-Switching Speech","ref_index":28,"is_internal_anchor":true},{"citing_arxiv_id":"2606.18699","citing_title":"TW-LegalBench: Measuring Taiwanese Legal Understanding","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EFNXFH6QZS64RMYW5C53RHPED2","json":"https://pith.science/pith/EFNXFH6QZS64RMYW5C53RHPED2.json","graph_json":"https://pith.science/api/pith-number/EFNXFH6QZS64RMYW5C53RHPED2/graph.json","events_json":"https://pith.science/api/pith-number/EFNXFH6QZS64RMYW5C53RHPED2/events.json","paper":"https://pith.science/paper/EFNXFH6Q"},"agent_actions":{"view_html":"https://pith.science/pith/EFNXFH6QZS64RMYW5C53RHPED2","download_json":"https://pith.science/pith/EFNXFH6QZS64RMYW5C53RHPED2.json","view_paper":"https://pith.science/paper/EFNXFH6Q","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.20180&json=true","fetch_graph":"https://pith.science/api/pith-number/EFNXFH6QZS64RMYW5C53RHPED2/graph.json","fetch_events":"https://pith.science/api/pith-number/EFNXFH6QZS64RMYW5C53RHPED2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EFNXFH6QZS64RMYW5C53RHPED2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EFNXFH6QZS64RMYW5C53RHPED2/action/storage_attestation","attest_author":"https://pith.science/pith/EFNXFH6QZS64RMYW5C53RHPED2/action/author_attestation","sign_citation":"https://pith.science/pith/EFNXFH6QZS64RMYW5C53RHPED2/action/citation_signature","submit_replication":"https://pith.science/pith/EFNXFH6QZS64RMYW5C53RHPED2/action/replication_record"}},"created_at":"2026-07-05T08:02:15.325469+00:00","updated_at":"2026-07-05T08:02:15.325469+00:00"}