{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:NTSRCRUZ5SXMGXDO4FPMMK7MQI","short_pith_number":"pith:NTSRCRUZ","schema_version":"1.0","canonical_sha256":"6ce5114699ecaec35c6ee15ec62bec823fb640d78033d32552a28ecfcf81cd02","source":{"kind":"arxiv","id":"2308.13149","version":2},"attestation_state":"computed","paper":{"title":"SciEval: A Multi-Level Large Language Model Evaluation Benchmark for Scientific Research","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Baocai Chen, Da Ma, Kai Yu, Liangtai Sun, Lu Chen, Yang Han, Zhennan Shen, Zihan Zhao","submitted_at":"2023-08-25T03:05:33Z","abstract_excerpt":"Recently, there has been growing interest in using Large Language Models (LLMs) for scientific research. Numerous benchmarks have been proposed to evaluate the ability of LLMs for scientific research. However, current benchmarks are mostly based on pre-collected objective questions. This design suffers from data leakage problem and lacks the evaluation of subjective Q/A ability. In this paper, we propose SciEval, a comprehensive and multi-disciplinary evaluation benchmark to address these issues. Based on Bloom's taxonomy, SciEval covers four dimensions to systematically evaluate scientific re"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.13149","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-08-25T03:05:33Z","cross_cats_sorted":[],"title_canon_sha256":"e19430da808b0f0b0b6758155d7627fb2c8ffd3ba0d4a7774adb410d5520ca29","abstract_canon_sha256":"91b923ca7607c3be0f68f2ba83a6892f522c338db0795d20e96138576b7faa44"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:32:03.343492Z","signature_b64":"tegM/EADAasCRQ9ejXlb6xa/MFohRfQSiJUFJwP5loS4eQwFLczdo7KAfrd4174U+EQhLoYxUKQp+rqCBkDGAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6ce5114699ecaec35c6ee15ec62bec823fb640d78033d32552a28ecfcf81cd02","last_reissued_at":"2026-07-05T09:32:03.342993Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:32:03.342993Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SciEval: A Multi-Level Large Language Model Evaluation Benchmark for Scientific Research","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Baocai Chen, Da Ma, Kai Yu, Liangtai Sun, Lu Chen, Yang Han, Zhennan Shen, Zihan Zhao","submitted_at":"2023-08-25T03:05:33Z","abstract_excerpt":"Recently, there has been growing interest in using Large Language Models (LLMs) for scientific research. Numerous benchmarks have been proposed to evaluate the ability of LLMs for scientific research. However, current benchmarks are mostly based on pre-collected objective questions. This design suffers from data leakage problem and lacks the evaluation of subjective Q/A ability. In this paper, we propose SciEval, a comprehensive and multi-disciplinary evaluation benchmark to address these issues. Based on Bloom's taxonomy, SciEval covers four dimensions to systematically evaluate scientific re"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.13149","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.13149/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.13149","created_at":"2026-07-05T09:32:03.343057+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.13149v2","created_at":"2026-07-05T09:32:03.343057+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.13149","created_at":"2026-07-05T09:32:03.343057+00:00"},{"alias_kind":"pith_short_12","alias_value":"NTSRCRUZ5SXM","created_at":"2026-07-05T09:32:03.343057+00:00"},{"alias_kind":"pith_short_16","alias_value":"NTSRCRUZ5SXMGXDO","created_at":"2026-07-05T09:32:03.343057+00:00"},{"alias_kind":"pith_short_8","alias_value":"NTSRCRUZ","created_at":"2026-07-05T09:32:03.343057+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2410.14702","citing_title":"Polymath: A Challenging Multi-modal Mathematical Reasoning Benchmark","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18630","citing_title":"SCICONVBENCH: Benchmarking LLMs on Multi-Turn Clarification for Task Formulation in Computational Science","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09554","citing_title":"LABBench2: An Improved Benchmark for AI Systems Performing Biology Research","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17966","citing_title":"TPS-CalcBench: A Benchmark and Diagnostic Evaluation Framework for LLM Analytical Calculation Competence in Hypersonic Thermal Protection System Engineering","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NTSRCRUZ5SXMGXDO4FPMMK7MQI","json":"https://pith.science/pith/NTSRCRUZ5SXMGXDO4FPMMK7MQI.json","graph_json":"https://pith.science/api/pith-number/NTSRCRUZ5SXMGXDO4FPMMK7MQI/graph.json","events_json":"https://pith.science/api/pith-number/NTSRCRUZ5SXMGXDO4FPMMK7MQI/events.json","paper":"https://pith.science/paper/NTSRCRUZ"},"agent_actions":{"view_html":"https://pith.science/pith/NTSRCRUZ5SXMGXDO4FPMMK7MQI","download_json":"https://pith.science/pith/NTSRCRUZ5SXMGXDO4FPMMK7MQI.json","view_paper":"https://pith.science/paper/NTSRCRUZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.13149&json=true","fetch_graph":"https://pith.science/api/pith-number/NTSRCRUZ5SXMGXDO4FPMMK7MQI/graph.json","fetch_events":"https://pith.science/api/pith-number/NTSRCRUZ5SXMGXDO4FPMMK7MQI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NTSRCRUZ5SXMGXDO4FPMMK7MQI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NTSRCRUZ5SXMGXDO4FPMMK7MQI/action/storage_attestation","attest_author":"https://pith.science/pith/NTSRCRUZ5SXMGXDO4FPMMK7MQI/action/author_attestation","sign_citation":"https://pith.science/pith/NTSRCRUZ5SXMGXDO4FPMMK7MQI/action/citation_signature","submit_replication":"https://pith.science/pith/NTSRCRUZ5SXMGXDO4FPMMK7MQI/action/replication_record"}},"created_at":"2026-07-05T09:32:03.343057+00:00","updated_at":"2026-07-05T09:32:03.343057+00:00"}