{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:PDBPBBTSEUNAJHBVYK6XLJMU2U","short_pith_number":"pith:PDBPBBTS","schema_version":"1.0","canonical_sha256":"78c2f08672251a049c35c2bd75a594d5070a89bb354e26736b0d50e572ae70d3","source":{"kind":"arxiv","id":"2504.20673","version":1},"attestation_state":"computed","paper":{"title":"CoCo-Bench: A Comprehensive Code Benchmark For Multi-task Large Language Model Evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SE","authors_text":"Da Lei, Dong Cao, Guangyao Su, Hao Chen, Jiancheng Wang, Jiawei Fang, Ji Pei, Menghang Dong, Peng Luo, Ran Chen, Shuai Yuan, Tianze Sun, Weifeng Liu, Wei Wang, Wenjing Yin, Xiang Ma, Yajun Zhang, Yijiong Yu, Yong Liu, Yuanjian Xu, Zekun Wang, Ziyun Dai","submitted_at":"2025-04-29T11:57:23Z","abstract_excerpt":"Large language models (LLMs) play a crucial role in software engineering, excelling in tasks like code generation and maintenance. However, existing benchmarks are often narrow in scope, focusing on a specific task and lack a comprehensive evaluation framework that reflects real-world applications. To address these gaps, we introduce CoCo-Bench (Comprehensive Code Benchmark), designed to evaluate LLMs across four critical dimensions: code understanding, code generation, code modification, and code review. These dimensions capture essential developer needs, ensuring a more systematic and repres"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.20673","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2025-04-29T11:57:23Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"963f892ecb7ee282ae396539409303fdb029935db00ef8a3f8b657445c86908b","abstract_canon_sha256":"e7d9c4861e173cbd1d475f660a61add7d021597bbe1acd425bbd31b865e3ea15"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:55:46.875677Z","signature_b64":"SDluJOLWcmrS69fdw+UinFvd5WajrvokMKG4Q9hoTfLEl9ffFzlTipVdAAONYQRgLQSeVyp8wZ43adtcSOa2DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"78c2f08672251a049c35c2bd75a594d5070a89bb354e26736b0d50e572ae70d3","last_reissued_at":"2026-07-05T10:55:46.875168Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:55:46.875168Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CoCo-Bench: A Comprehensive Code Benchmark For Multi-task Large Language Model Evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SE","authors_text":"Da Lei, Dong Cao, Guangyao Su, Hao Chen, Jiancheng Wang, Jiawei Fang, Ji Pei, Menghang Dong, Peng Luo, Ran Chen, Shuai Yuan, Tianze Sun, Weifeng Liu, Wei Wang, Wenjing Yin, Xiang Ma, Yajun Zhang, Yijiong Yu, Yong Liu, Yuanjian Xu, Zekun Wang, Ziyun Dai","submitted_at":"2025-04-29T11:57:23Z","abstract_excerpt":"Large language models (LLMs) play a crucial role in software engineering, excelling in tasks like code generation and maintenance. However, existing benchmarks are often narrow in scope, focusing on a specific task and lack a comprehensive evaluation framework that reflects real-world applications. To address these gaps, we introduce CoCo-Bench (Comprehensive Code Benchmark), designed to evaluate LLMs across four critical dimensions: code understanding, code generation, code modification, and code review. These dimensions capture essential developer needs, ensuring a more systematic and repres"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.20673","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.20673/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.20673","created_at":"2026-07-05T10:55:46.875235+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.20673v1","created_at":"2026-07-05T10:55:46.875235+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.20673","created_at":"2026-07-05T10:55:46.875235+00:00"},{"alias_kind":"pith_short_12","alias_value":"PDBPBBTSEUNA","created_at":"2026-07-05T10:55:46.875235+00:00"},{"alias_kind":"pith_short_16","alias_value":"PDBPBBTSEUNAJHBV","created_at":"2026-07-05T10:55:46.875235+00:00"},{"alias_kind":"pith_short_8","alias_value":"PDBPBBTS","created_at":"2026-07-05T10:55:46.875235+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PDBPBBTSEUNAJHBVYK6XLJMU2U","json":"https://pith.science/pith/PDBPBBTSEUNAJHBVYK6XLJMU2U.json","graph_json":"https://pith.science/api/pith-number/PDBPBBTSEUNAJHBVYK6XLJMU2U/graph.json","events_json":"https://pith.science/api/pith-number/PDBPBBTSEUNAJHBVYK6XLJMU2U/events.json","paper":"https://pith.science/paper/PDBPBBTS"},"agent_actions":{"view_html":"https://pith.science/pith/PDBPBBTSEUNAJHBVYK6XLJMU2U","download_json":"https://pith.science/pith/PDBPBBTSEUNAJHBVYK6XLJMU2U.json","view_paper":"https://pith.science/paper/PDBPBBTS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.20673&json=true","fetch_graph":"https://pith.science/api/pith-number/PDBPBBTSEUNAJHBVYK6XLJMU2U/graph.json","fetch_events":"https://pith.science/api/pith-number/PDBPBBTSEUNAJHBVYK6XLJMU2U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PDBPBBTSEUNAJHBVYK6XLJMU2U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PDBPBBTSEUNAJHBVYK6XLJMU2U/action/storage_attestation","attest_author":"https://pith.science/pith/PDBPBBTSEUNAJHBVYK6XLJMU2U/action/author_attestation","sign_citation":"https://pith.science/pith/PDBPBBTSEUNAJHBVYK6XLJMU2U/action/citation_signature","submit_replication":"https://pith.science/pith/PDBPBBTSEUNAJHBVYK6XLJMU2U/action/replication_record"}},"created_at":"2026-07-05T10:55:46.875235+00:00","updated_at":"2026-07-05T10:55:46.875235+00:00"}