{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:Q5LE5ILCYBP5BGRYQWCCXMT2SI","short_pith_number":"pith:Q5LE5ILC","schema_version":"1.0","canonical_sha256":"87564ea162c05fd09a3885842bb27a9212bddde6440cbc6831a956927b300832","source":{"kind":"arxiv","id":"2505.16591","version":1},"attestation_state":"computed","paper":{"title":"Evaluating Large Language Model with Knowledge Oriented Language Specific Simple Question Answering","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bowen Jiang, Conghui He, Dahua Lin, Haote Yang, Jiang Wu, Jia Yu, Junyuan Gao, Lijun Wu, Rui Min, Runchuan Zhu, Songyang Zhang, Yifan He, Yinfan Wang, Zinco Jiang","submitted_at":"2025-05-22T12:27:02Z","abstract_excerpt":"We introduce KoLasSimpleQA, the first benchmark evaluating the multilingual factual ability of Large Language Models (LLMs). Inspired by existing research, we created the question set with features such as single knowledge point coverage, absolute objectivity, unique answers, and temporal stability. These questions enable efficient evaluation using the LLM-as-judge paradigm, testing both the LLMs' factual memory and self-awareness (\"know what they don't know\"). KoLasSimpleQA expands existing research in two key dimensions: (1) Breadth (Multilingual Coverage): It includes 9 languages, supportin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.16591","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-22T12:27:02Z","cross_cats_sorted":[],"title_canon_sha256":"715fbfe85ac1c431683b8285e79d2718fd1a6f41fe980b8fa734d58b89e207c9","abstract_canon_sha256":"e1dbaa1412d79b0bcd5e1a1a09d2eb98ad23bf506f0dbf4071865f42145325c2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:07:37.568146Z","signature_b64":"UhayOf7Fd/i76Rj1wX3WQddbf4VQcZF4JNvOSRMXqmLivsSBZjrkUeYBGLfzH70GnKJnCW1+YvFI82qK9EV7Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"87564ea162c05fd09a3885842bb27a9212bddde6440cbc6831a956927b300832","last_reissued_at":"2026-07-05T11:07:37.567563Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:07:37.567563Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating Large Language Model with Knowledge Oriented Language Specific Simple Question Answering","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bowen Jiang, Conghui He, Dahua Lin, Haote Yang, Jiang Wu, Jia Yu, Junyuan Gao, Lijun Wu, Rui Min, Runchuan Zhu, Songyang Zhang, Yifan He, Yinfan Wang, Zinco Jiang","submitted_at":"2025-05-22T12:27:02Z","abstract_excerpt":"We introduce KoLasSimpleQA, the first benchmark evaluating the multilingual factual ability of Large Language Models (LLMs). Inspired by existing research, we created the question set with features such as single knowledge point coverage, absolute objectivity, unique answers, and temporal stability. These questions enable efficient evaluation using the LLM-as-judge paradigm, testing both the LLMs' factual memory and self-awareness (\"know what they don't know\"). KoLasSimpleQA expands existing research in two key dimensions: (1) Breadth (Multilingual Coverage): It includes 9 languages, supportin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.16591","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.16591/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.16591","created_at":"2026-07-05T11:07:37.567629+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.16591v1","created_at":"2026-07-05T11:07:37.567629+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.16591","created_at":"2026-07-05T11:07:37.567629+00:00"},{"alias_kind":"pith_short_12","alias_value":"Q5LE5ILCYBP5","created_at":"2026-07-05T11:07:37.567629+00:00"},{"alias_kind":"pith_short_16","alias_value":"Q5LE5ILCYBP5BGRY","created_at":"2026-07-05T11:07:37.567629+00:00"},{"alias_kind":"pith_short_8","alias_value":"Q5LE5ILC","created_at":"2026-07-05T11:07:37.567629+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Q5LE5ILCYBP5BGRYQWCCXMT2SI","json":"https://pith.science/pith/Q5LE5ILCYBP5BGRYQWCCXMT2SI.json","graph_json":"https://pith.science/api/pith-number/Q5LE5ILCYBP5BGRYQWCCXMT2SI/graph.json","events_json":"https://pith.science/api/pith-number/Q5LE5ILCYBP5BGRYQWCCXMT2SI/events.json","paper":"https://pith.science/paper/Q5LE5ILC"},"agent_actions":{"view_html":"https://pith.science/pith/Q5LE5ILCYBP5BGRYQWCCXMT2SI","download_json":"https://pith.science/pith/Q5LE5ILCYBP5BGRYQWCCXMT2SI.json","view_paper":"https://pith.science/paper/Q5LE5ILC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.16591&json=true","fetch_graph":"https://pith.science/api/pith-number/Q5LE5ILCYBP5BGRYQWCCXMT2SI/graph.json","fetch_events":"https://pith.science/api/pith-number/Q5LE5ILCYBP5BGRYQWCCXMT2SI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Q5LE5ILCYBP5BGRYQWCCXMT2SI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Q5LE5ILCYBP5BGRYQWCCXMT2SI/action/storage_attestation","attest_author":"https://pith.science/pith/Q5LE5ILCYBP5BGRYQWCCXMT2SI/action/author_attestation","sign_citation":"https://pith.science/pith/Q5LE5ILCYBP5BGRYQWCCXMT2SI/action/citation_signature","submit_replication":"https://pith.science/pith/Q5LE5ILCYBP5BGRYQWCCXMT2SI/action/replication_record"}},"created_at":"2026-07-05T11:07:37.567629+00:00","updated_at":"2026-07-05T11:07:37.567629+00:00"}