{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:BQKYUWFIAGRL6U2NQJ5X3MDKCT","short_pith_number":"pith:BQKYUWFI","schema_version":"1.0","canonical_sha256":"0c158a58a801a2bf534d827b7db06a14ff2053803b7b78e672fa6a2efb5ea948","source":{"kind":"arxiv","id":"2502.16614","version":1},"attestation_state":"computed","paper":{"title":"CodeCriticBench: A Holistic Code Critique Benchmark for Large Language Models","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alexander Zhang, Ge Zhang, Jiaheng Liu, Jian Yang, Ken Deng, Marcus Dong, Tianyu Liu, Wangchunshu Zhou, Weixun Wang, Wei Zhang, Wenhao Huang, Yancheng He, Yejie Wang, Yingshui Tan, Yuanxing Zhang, Zhaoxiang Zhang, Zhexu Wang, Zhongyuan Peng","submitted_at":"2025-02-23T15:36:43Z","abstract_excerpt":"The critique capacity of Large Language Models (LLMs) is essential for reasoning abilities, which can provide necessary suggestions (e.g., detailed analysis and constructive feedback). Therefore, how to evaluate the critique capacity of LLMs has drawn great attention and several critique benchmarks have been proposed. However, existing critique benchmarks usually have the following limitations: (1). Focusing on diverse reasoning tasks in general domains and insufficient evaluation on code tasks (e.g., only covering code generation task), where the difficulty of queries is relatively easy (e.g."},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.16614","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.CL","submitted_at":"2025-02-23T15:36:43Z","cross_cats_sorted":[],"title_canon_sha256":"38d5e3e55c5bbd8f96ee711ba11ffb65c300f212fef7ad399bed823466d78855","abstract_canon_sha256":"da54e662a5076b76d284218188431017805abe576c184eff61c08ae610fd22f9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:18:56.413587Z","signature_b64":"EtCXHwSc0rvSw1WweYdrd4f5WtfWDUWZJ3sDoHb/i+ByPPRa+nk2Tk7zbq/NMUO1JGuyCA//50pHsU+fZYznAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0c158a58a801a2bf534d827b7db06a14ff2053803b7b78e672fa6a2efb5ea948","last_reissued_at":"2026-07-05T10:18:56.413098Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:18:56.413098Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CodeCriticBench: A Holistic Code Critique Benchmark for Large Language Models","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alexander Zhang, Ge Zhang, Jiaheng Liu, Jian Yang, Ken Deng, Marcus Dong, Tianyu Liu, Wangchunshu Zhou, Weixun Wang, Wei Zhang, Wenhao Huang, Yancheng He, Yejie Wang, Yingshui Tan, Yuanxing Zhang, Zhaoxiang Zhang, Zhexu Wang, Zhongyuan Peng","submitted_at":"2025-02-23T15:36:43Z","abstract_excerpt":"The critique capacity of Large Language Models (LLMs) is essential for reasoning abilities, which can provide necessary suggestions (e.g., detailed analysis and constructive feedback). Therefore, how to evaluate the critique capacity of LLMs has drawn great attention and several critique benchmarks have been proposed. However, existing critique benchmarks usually have the following limitations: (1). Focusing on diverse reasoning tasks in general domains and insufficient evaluation on code tasks (e.g., only covering code generation task), where the difficulty of queries is relatively easy (e.g."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.16614","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.16614/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.16614","created_at":"2026-07-05T10:18:56.413149+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.16614v1","created_at":"2026-07-05T10:18:56.413149+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.16614","created_at":"2026-07-05T10:18:56.413149+00:00"},{"alias_kind":"pith_short_12","alias_value":"BQKYUWFIAGRL","created_at":"2026-07-05T10:18:56.413149+00:00"},{"alias_kind":"pith_short_16","alias_value":"BQKYUWFIAGRL6U2N","created_at":"2026-07-05T10:18:56.413149+00:00"},{"alias_kind":"pith_short_8","alias_value":"BQKYUWFI","created_at":"2026-07-05T10:18:56.413149+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.15079","citing_title":"Ling and Ring 2.6 Technical Report: Efficient and Instant Agentic Intelligence at Trillion-Parameter Scale","ref_index":173,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05238","citing_title":"DeployBench: Benchmarking LLM Agents for Research Artifact Deployment","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06840","citing_title":"Characterize Then Distill: Mechanistic Reasoning in Large Output Spaces","ref_index":116,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02801","citing_title":"Reinforcement Learning for LLM-based Multi-Agent Systems through Orchestration Traces","ref_index":83,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BQKYUWFIAGRL6U2NQJ5X3MDKCT","json":"https://pith.science/pith/BQKYUWFIAGRL6U2NQJ5X3MDKCT.json","graph_json":"https://pith.science/api/pith-number/BQKYUWFIAGRL6U2NQJ5X3MDKCT/graph.json","events_json":"https://pith.science/api/pith-number/BQKYUWFIAGRL6U2NQJ5X3MDKCT/events.json","paper":"https://pith.science/paper/BQKYUWFI"},"agent_actions":{"view_html":"https://pith.science/pith/BQKYUWFIAGRL6U2NQJ5X3MDKCT","download_json":"https://pith.science/pith/BQKYUWFIAGRL6U2NQJ5X3MDKCT.json","view_paper":"https://pith.science/paper/BQKYUWFI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.16614&json=true","fetch_graph":"https://pith.science/api/pith-number/BQKYUWFIAGRL6U2NQJ5X3MDKCT/graph.json","fetch_events":"https://pith.science/api/pith-number/BQKYUWFIAGRL6U2NQJ5X3MDKCT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BQKYUWFIAGRL6U2NQJ5X3MDKCT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BQKYUWFIAGRL6U2NQJ5X3MDKCT/action/storage_attestation","attest_author":"https://pith.science/pith/BQKYUWFIAGRL6U2NQJ5X3MDKCT/action/author_attestation","sign_citation":"https://pith.science/pith/BQKYUWFIAGRL6U2NQJ5X3MDKCT/action/citation_signature","submit_replication":"https://pith.science/pith/BQKYUWFIAGRL6U2NQJ5X3MDKCT/action/replication_record"}},"created_at":"2026-07-05T10:18:56.413149+00:00","updated_at":"2026-07-05T10:18:56.413149+00:00"}