{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DIAMFSQ26CRV2255I4COHMCJCK","short_pith_number":"pith:DIAMFSQ2","schema_version":"1.0","canonical_sha256":"1a00c2ca1af0a35d6bbd4704e3b04912aa600ef92faa56c64ab258756f52266e","source":{"kind":"arxiv","id":"2504.13216","version":1},"attestation_state":"computed","paper":{"title":"KFinEval-Pilot: A Comprehensive Benchmark Suite for Korean Financial Language Understanding","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Bokwang Hwang, Chanhyuk Yoon, Chansu Lee, Hangyeol Yoo, Heeyewon Jeong, Hyebin Kang, Jihyun Park, Jina Park, Jingyeong Hong, Jinsun Yoo, Jinwoo Lee, Jiyeon Lee, KyungTae Lim, Myeonggyu Lee, Seonhye Gu, Seonkyu Lim, SoHyun Park, Suhyun Kim, Sunghyun Bang, Taewoong Kim, Yejee Kang, Yerin Kim, Yiseul Lee, Yongchan Kim, Yongjae Geun, Younggyun Hahm, Yousang Cho","submitted_at":"2025-04-17T00:12:58Z","abstract_excerpt":"We introduce KFinEval-Pilot, a benchmark suite specifically designed to evaluate large language models (LLMs) in the Korean financial domain. Addressing the limitations of existing English-centric benchmarks, KFinEval-Pilot comprises over 1,000 curated questions across three critical areas: financial knowledge, legal reasoning, and financial toxicity. The benchmark is constructed through a semi-automated pipeline that combines GPT-4-generated prompts with expert validation to ensure domain relevance and factual accuracy. We evaluate a range of representative LLMs and observe notable performanc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.13216","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2025-04-17T00:12:58Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"aa9c4c3e3614c082c466faafbeb20cd7b458fd3b6b145818ef26a91db2be6f6c","abstract_canon_sha256":"dc0a27e8bd6745fc90607ef8a1aad1248fe091ea42cc964aec6df839c294e648"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:50:50.766259Z","signature_b64":"IH9mReQK3naFwQ0TKvmOMx0+shsd5HsoXZuB6rovZEY9SF50FqiYWN3Di//iWjc6x380ubftddwaFeeLAWrwBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1a00c2ca1af0a35d6bbd4704e3b04912aa600ef92faa56c64ab258756f52266e","last_reissued_at":"2026-07-05T10:50:50.765672Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:50:50.765672Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"KFinEval-Pilot: A Comprehensive Benchmark Suite for Korean Financial Language Understanding","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Bokwang Hwang, Chanhyuk Yoon, Chansu Lee, Hangyeol Yoo, Heeyewon Jeong, Hyebin Kang, Jihyun Park, Jina Park, Jingyeong Hong, Jinsun Yoo, Jinwoo Lee, Jiyeon Lee, KyungTae Lim, Myeonggyu Lee, Seonhye Gu, Seonkyu Lim, SoHyun Park, Suhyun Kim, Sunghyun Bang, Taewoong Kim, Yejee Kang, Yerin Kim, Yiseul Lee, Yongchan Kim, Yongjae Geun, Younggyun Hahm, Yousang Cho","submitted_at":"2025-04-17T00:12:58Z","abstract_excerpt":"We introduce KFinEval-Pilot, a benchmark suite specifically designed to evaluate large language models (LLMs) in the Korean financial domain. Addressing the limitations of existing English-centric benchmarks, KFinEval-Pilot comprises over 1,000 curated questions across three critical areas: financial knowledge, legal reasoning, and financial toxicity. The benchmark is constructed through a semi-automated pipeline that combines GPT-4-generated prompts with expert validation to ensure domain relevance and factual accuracy. We evaluate a range of representative LLMs and observe notable performanc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.13216","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.13216/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.13216","created_at":"2026-07-05T10:50:50.765744+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.13216v1","created_at":"2026-07-05T10:50:50.765744+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.13216","created_at":"2026-07-05T10:50:50.765744+00:00"},{"alias_kind":"pith_short_12","alias_value":"DIAMFSQ26CRV","created_at":"2026-07-05T10:50:50.765744+00:00"},{"alias_kind":"pith_short_16","alias_value":"DIAMFSQ26CRV2255","created_at":"2026-07-05T10:50:50.765744+00:00"},{"alias_kind":"pith_short_8","alias_value":"DIAMFSQ2","created_at":"2026-07-05T10:50:50.765744+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DIAMFSQ26CRV2255I4COHMCJCK","json":"https://pith.science/pith/DIAMFSQ26CRV2255I4COHMCJCK.json","graph_json":"https://pith.science/api/pith-number/DIAMFSQ26CRV2255I4COHMCJCK/graph.json","events_json":"https://pith.science/api/pith-number/DIAMFSQ26CRV2255I4COHMCJCK/events.json","paper":"https://pith.science/paper/DIAMFSQ2"},"agent_actions":{"view_html":"https://pith.science/pith/DIAMFSQ26CRV2255I4COHMCJCK","download_json":"https://pith.science/pith/DIAMFSQ26CRV2255I4COHMCJCK.json","view_paper":"https://pith.science/paper/DIAMFSQ2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.13216&json=true","fetch_graph":"https://pith.science/api/pith-number/DIAMFSQ26CRV2255I4COHMCJCK/graph.json","fetch_events":"https://pith.science/api/pith-number/DIAMFSQ26CRV2255I4COHMCJCK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DIAMFSQ26CRV2255I4COHMCJCK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DIAMFSQ26CRV2255I4COHMCJCK/action/storage_attestation","attest_author":"https://pith.science/pith/DIAMFSQ26CRV2255I4COHMCJCK/action/author_attestation","sign_citation":"https://pith.science/pith/DIAMFSQ26CRV2255I4COHMCJCK/action/citation_signature","submit_replication":"https://pith.science/pith/DIAMFSQ26CRV2255I4COHMCJCK/action/replication_record"}},"created_at":"2026-07-05T10:50:50.765744+00:00","updated_at":"2026-07-05T10:50:50.765744+00:00"}