{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XBJ4ELSO5YIK43NBKQJSX3RRBA","short_pith_number":"pith:XBJ4ELSO","schema_version":"1.0","canonical_sha256":"b853c22e4eee10ae6da154132bee31081884680f26d89ce21e836d2efaabea51","source":{"kind":"arxiv","id":"2403.01976","version":5},"attestation_state":"computed","paper":{"title":"SciAssess: Benchmarking LLM Proficiency in Scientific Literature Analysis","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Changhong Chen, Changxin Wang, Guolin Ke, Hengxing Cai, Hongshuai Wang, Jiankun Wang, Jiaxi Zhuang, Jin Huang, Junhan Chang, Linfeng Zhang, Lin Yao, Mingjun Xu, Mujie Lin, Shuwen Yang, Sihang Li, Xiaochen Cai, Xi Fang, Yaqi Li, Yongge Li, Yuqi Yin, Zheng Cheng, Zhifeng Gao, Zifeng Zhao","submitted_at":"2024-03-04T12:19:28Z","abstract_excerpt":"Recent breakthroughs in Large Language Models (LLMs) have revolutionized scientific literature analysis. However, existing benchmarks fail to adequately evaluate the proficiency of LLMs in this domain, particularly in scenarios requiring higher-level abilities beyond mere memorization and the handling of multimodal data. In response to this gap, we introduce SciAssess, a benchmark specifically designed for the comprehensive evaluation of LLMs in scientific literature analysis. It aims to thoroughly assess the efficacy of LLMs by evaluating their capabilities in Memorization (L1), Comprehension"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.01976","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-03-04T12:19:28Z","cross_cats_sorted":[],"title_canon_sha256":"1e733c95b66e3b2151958e0c1194696577b941085ddbde92b95bd72f88a8d2c2","abstract_canon_sha256":"8ef9f6a167aade8b48e3992f804fab7493939c40bb6fbe7c4bbeca5c20552751"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:22:16.363349Z","signature_b64":"KGwX2NzFn7jigK223byDYu8Fih/W84iG3YSsqAB2S21iN/Tokj9Ty1z6xLY6z+yyve70t9dLHo2ka3wVGFqdBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b853c22e4eee10ae6da154132bee31081884680f26d89ce21e836d2efaabea51","last_reissued_at":"2026-07-05T09:22:16.362838Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:22:16.362838Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SciAssess: Benchmarking LLM Proficiency in Scientific Literature Analysis","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Changhong Chen, Changxin Wang, Guolin Ke, Hengxing Cai, Hongshuai Wang, Jiankun Wang, Jiaxi Zhuang, Jin Huang, Junhan Chang, Linfeng Zhang, Lin Yao, Mingjun Xu, Mujie Lin, Shuwen Yang, Sihang Li, Xiaochen Cai, Xi Fang, Yaqi Li, Yongge Li, Yuqi Yin, Zheng Cheng, Zhifeng Gao, Zifeng Zhao","submitted_at":"2024-03-04T12:19:28Z","abstract_excerpt":"Recent breakthroughs in Large Language Models (LLMs) have revolutionized scientific literature analysis. However, existing benchmarks fail to adequately evaluate the proficiency of LLMs in this domain, particularly in scenarios requiring higher-level abilities beyond mere memorization and the handling of multimodal data. In response to this gap, we introduce SciAssess, a benchmark specifically designed for the comprehensive evaluation of LLMs in scientific literature analysis. It aims to thoroughly assess the efficacy of LLMs by evaluating their capabilities in Memorization (L1), Comprehension"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.01976","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.01976/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.01976","created_at":"2026-07-05T09:22:16.362897+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.01976v5","created_at":"2026-07-05T09:22:16.362897+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.01976","created_at":"2026-07-05T09:22:16.362897+00:00"},{"alias_kind":"pith_short_12","alias_value":"XBJ4ELSO5YIK","created_at":"2026-07-05T09:22:16.362897+00:00"},{"alias_kind":"pith_short_16","alias_value":"XBJ4ELSO5YIK43NB","created_at":"2026-07-05T09:22:16.362897+00:00"},{"alias_kind":"pith_short_8","alias_value":"XBJ4ELSO","created_at":"2026-07-05T09:22:16.362897+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20897","citing_title":"PeerCheck: Enhancing LLM-Generated Academic Reviews Towards Human-Level Quality","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17971","citing_title":"Babel: Jailbreaking Safety Attention via Obfuscation Distribution Optimized Sampling","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XBJ4ELSO5YIK43NBKQJSX3RRBA","json":"https://pith.science/pith/XBJ4ELSO5YIK43NBKQJSX3RRBA.json","graph_json":"https://pith.science/api/pith-number/XBJ4ELSO5YIK43NBKQJSX3RRBA/graph.json","events_json":"https://pith.science/api/pith-number/XBJ4ELSO5YIK43NBKQJSX3RRBA/events.json","paper":"https://pith.science/paper/XBJ4ELSO"},"agent_actions":{"view_html":"https://pith.science/pith/XBJ4ELSO5YIK43NBKQJSX3RRBA","download_json":"https://pith.science/pith/XBJ4ELSO5YIK43NBKQJSX3RRBA.json","view_paper":"https://pith.science/paper/XBJ4ELSO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.01976&json=true","fetch_graph":"https://pith.science/api/pith-number/XBJ4ELSO5YIK43NBKQJSX3RRBA/graph.json","fetch_events":"https://pith.science/api/pith-number/XBJ4ELSO5YIK43NBKQJSX3RRBA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XBJ4ELSO5YIK43NBKQJSX3RRBA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XBJ4ELSO5YIK43NBKQJSX3RRBA/action/storage_attestation","attest_author":"https://pith.science/pith/XBJ4ELSO5YIK43NBKQJSX3RRBA/action/author_attestation","sign_citation":"https://pith.science/pith/XBJ4ELSO5YIK43NBKQJSX3RRBA/action/citation_signature","submit_replication":"https://pith.science/pith/XBJ4ELSO5YIK43NBKQJSX3RRBA/action/replication_record"}},"created_at":"2026-07-05T09:22:16.362897+00:00","updated_at":"2026-07-05T09:22:16.362897+00:00"}