{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:FHFB6XWOHDLJHW36YSA6K3GHLU","short_pith_number":"pith:FHFB6XWO","schema_version":"1.0","canonical_sha256":"29ca1f5ece38d693db7ec481e56cc75d3c82d05ad7b3fdc634506628d8ab3198","source":{"kind":"arxiv","id":"2204.04541","version":1},"attestation_state":"computed","paper":{"title":"KOBEST: Korean Balanced Evaluation of Significant Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Deuk Sin Kwon, Dohyeong Kim, Eric Davis, Myeongjun Jang","submitted_at":"2022-04-09T20:13:51Z","abstract_excerpt":"A well-formulated benchmark plays a critical role in spurring advancements in the natural language processing (NLP) field, as it allows objective and precise evaluation of diverse models. As modern language models (LMs) have become more elaborate and sophisticated, more difficult benchmarks that require linguistic knowledge and reasoning have been proposed. However, most of these benchmarks only support English, and great effort is necessary to construct benchmarks for other low resource languages. To this end, we propose a new benchmark named Korean balanced evaluation of significant tasks (K"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2204.04541","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-04-09T20:13:51Z","cross_cats_sorted":[],"title_canon_sha256":"25bbae054d2db5e596ce17bdce3e6223d28e9a3bb375448f0b3ec9abe8a3f9b1","abstract_canon_sha256":"8dbe868998f7f5f1b7ae65212aa8d72cf09e946b68d9f11c928d6adad6627a7a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:12:58.381370Z","signature_b64":"YfAAaiOITcXu9p6sM6U8vlRMegWxiGZx4d79O17GAdMAVXgNjXX9DvlDLquVRX16WcEHqD/lUHg9+canOEW3AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"29ca1f5ece38d693db7ec481e56cc75d3c82d05ad7b3fdc634506628d8ab3198","last_reissued_at":"2026-07-05T04:12:58.381005Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:12:58.381005Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"KOBEST: Korean Balanced Evaluation of Significant Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Deuk Sin Kwon, Dohyeong Kim, Eric Davis, Myeongjun Jang","submitted_at":"2022-04-09T20:13:51Z","abstract_excerpt":"A well-formulated benchmark plays a critical role in spurring advancements in the natural language processing (NLP) field, as it allows objective and precise evaluation of diverse models. As modern language models (LMs) have become more elaborate and sophisticated, more difficult benchmarks that require linguistic knowledge and reasoning have been proposed. However, most of these benchmarks only support English, and great effort is necessary to construct benchmarks for other low resource languages. To this end, we propose a new benchmark named Korean balanced evaluation of significant tasks (K"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2204.04541","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2204.04541/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2204.04541","created_at":"2026-07-05T04:12:58.381060+00:00"},{"alias_kind":"arxiv_version","alias_value":"2204.04541v1","created_at":"2026-07-05T04:12:58.381060+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2204.04541","created_at":"2026-07-05T04:12:58.381060+00:00"},{"alias_kind":"pith_short_12","alias_value":"FHFB6XWOHDLJ","created_at":"2026-07-05T04:12:58.381060+00:00"},{"alias_kind":"pith_short_16","alias_value":"FHFB6XWOHDLJHW36","created_at":"2026-07-05T04:12:58.381060+00:00"},{"alias_kind":"pith_short_8","alias_value":"FHFB6XWO","created_at":"2026-07-05T04:12:58.381060+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.01034","citing_title":"Opt.Gear Technical Report","ref_index":21,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FHFB6XWOHDLJHW36YSA6K3GHLU","json":"https://pith.science/pith/FHFB6XWOHDLJHW36YSA6K3GHLU.json","graph_json":"https://pith.science/api/pith-number/FHFB6XWOHDLJHW36YSA6K3GHLU/graph.json","events_json":"https://pith.science/api/pith-number/FHFB6XWOHDLJHW36YSA6K3GHLU/events.json","paper":"https://pith.science/paper/FHFB6XWO"},"agent_actions":{"view_html":"https://pith.science/pith/FHFB6XWOHDLJHW36YSA6K3GHLU","download_json":"https://pith.science/pith/FHFB6XWOHDLJHW36YSA6K3GHLU.json","view_paper":"https://pith.science/paper/FHFB6XWO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2204.04541&json=true","fetch_graph":"https://pith.science/api/pith-number/FHFB6XWOHDLJHW36YSA6K3GHLU/graph.json","fetch_events":"https://pith.science/api/pith-number/FHFB6XWOHDLJHW36YSA6K3GHLU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FHFB6XWOHDLJHW36YSA6K3GHLU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FHFB6XWOHDLJHW36YSA6K3GHLU/action/storage_attestation","attest_author":"https://pith.science/pith/FHFB6XWOHDLJHW36YSA6K3GHLU/action/author_attestation","sign_citation":"https://pith.science/pith/FHFB6XWOHDLJHW36YSA6K3GHLU/action/citation_signature","submit_replication":"https://pith.science/pith/FHFB6XWOHDLJHW36YSA6K3GHLU/action/replication_record"}},"created_at":"2026-07-05T04:12:58.381060+00:00","updated_at":"2026-07-05T04:12:58.381060+00:00"}