{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:IEWBA3OEGRWZLXTMDPVAPATG3Y","short_pith_number":"pith:IEWBA3OE","schema_version":"1.0","canonical_sha256":"412c106dc4346d95de6c1bea078266de18686248f567448aa1f2f33f0125027d","source":{"kind":"arxiv","id":"2507.10535","version":2},"attestation_state":"computed","paper":{"title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.SE"],"primary_cat":"cs.CL","authors_text":"Hongchao Jiang, Hung-yi Lee, Robby T. Tan, Yiming Chen, Yushi Cao","submitted_at":"2025-07-14T17:56:29Z","abstract_excerpt":"Large Language Models (LLMs) have significantly advanced the state-of-the-art in various coding tasks. Beyond directly answering user queries, LLMs can also serve as judges, assessing and comparing the quality of responses generated by other models. Such an evaluation capability is crucial both for benchmarking different LLMs and for improving response quality through response ranking. However, despite the growing adoption of the LLM-as-a-Judge paradigm, its effectiveness in coding scenarios remains underexplored due to the absence of dedicated benchmarks. To address this gap, we introduce Cod"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.10535","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-07-14T17:56:29Z","cross_cats_sorted":["cs.AI","cs.SE"],"title_canon_sha256":"25e66b0309e43395b0dfb6017fcc6e30176e2941f0b8e664cba3541b203d65ec","abstract_canon_sha256":"48076e1e17b27a30fe8cabaca6c02e7590c7a2cb5e1fd7d2f4bcefda04a70474"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:53:48.843170Z","signature_b64":"7KSWahRVKvYdiW/63YHyh6Ik904B364lZ95AIRF8zxldWRJDIH4qjEHUYuKcoSjXkfK+UYIaLL+DNo/dPUenDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"412c106dc4346d95de6c1bea078266de18686248f567448aa1f2f33f0125027d","last_reissued_at":"2026-07-05T11:53:48.842591Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:53:48.842591Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.SE"],"primary_cat":"cs.CL","authors_text":"Hongchao Jiang, Hung-yi Lee, Robby T. Tan, Yiming Chen, Yushi Cao","submitted_at":"2025-07-14T17:56:29Z","abstract_excerpt":"Large Language Models (LLMs) have significantly advanced the state-of-the-art in various coding tasks. Beyond directly answering user queries, LLMs can also serve as judges, assessing and comparing the quality of responses generated by other models. Such an evaluation capability is crucial both for benchmarking different LLMs and for improving response quality through response ranking. However, despite the growing adoption of the LLM-as-a-Judge paradigm, its effectiveness in coding scenarios remains underexplored due to the absence of dedicated benchmarks. To address this gap, we introduce Cod"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.10535","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.10535/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.10535","created_at":"2026-07-05T11:53:48.842676+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.10535v2","created_at":"2026-07-05T11:53:48.842676+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.10535","created_at":"2026-07-05T11:53:48.842676+00:00"},{"alias_kind":"pith_short_12","alias_value":"IEWBA3OEGRWZ","created_at":"2026-07-05T11:53:48.842676+00:00"},{"alias_kind":"pith_short_16","alias_value":"IEWBA3OEGRWZLXTM","created_at":"2026-07-05T11:53:48.842676+00:00"},{"alias_kind":"pith_short_8","alias_value":"IEWBA3OE","created_at":"2026-07-05T11:53:48.842676+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05391","citing_title":"LLM-as-a-Verifier: A General-Purpose Verification Framework","ref_index":69,"is_internal_anchor":true},{"citing_arxiv_id":"2509.09936","citing_title":"SciML Agents: Write the Solver, Not the Solution","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13139","citing_title":"SWE-Cycle: Benchmarking Code Agents across the Complete Issue Resolution Cycle","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02906","citing_title":"OpsLLM: Construction of Large Language Model for Software Operations with Multi-stage Learning","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27727","citing_title":"LLM-as-a-Judge for Human-AI Co-Creation: A Reliability-Aware Evaluation Framework for Coding","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01474","citing_title":"ReMedi: Reasoner for Medical Clinical Prediction","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02906","citing_title":"OpsLLM: Construction of Large Language Model for Software Operations with Multi-stage Learning","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16790","citing_title":"Bias in the Loop: Auditing LLM-as-a-Judge for Software Engineering","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IEWBA3OEGRWZLXTMDPVAPATG3Y","json":"https://pith.science/pith/IEWBA3OEGRWZLXTMDPVAPATG3Y.json","graph_json":"https://pith.science/api/pith-number/IEWBA3OEGRWZLXTMDPVAPATG3Y/graph.json","events_json":"https://pith.science/api/pith-number/IEWBA3OEGRWZLXTMDPVAPATG3Y/events.json","paper":"https://pith.science/paper/IEWBA3OE"},"agent_actions":{"view_html":"https://pith.science/pith/IEWBA3OEGRWZLXTMDPVAPATG3Y","download_json":"https://pith.science/pith/IEWBA3OEGRWZLXTMDPVAPATG3Y.json","view_paper":"https://pith.science/paper/IEWBA3OE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.10535&json=true","fetch_graph":"https://pith.science/api/pith-number/IEWBA3OEGRWZLXTMDPVAPATG3Y/graph.json","fetch_events":"https://pith.science/api/pith-number/IEWBA3OEGRWZLXTMDPVAPATG3Y/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IEWBA3OEGRWZLXTMDPVAPATG3Y/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IEWBA3OEGRWZLXTMDPVAPATG3Y/action/storage_attestation","attest_author":"https://pith.science/pith/IEWBA3OEGRWZLXTMDPVAPATG3Y/action/author_attestation","sign_citation":"https://pith.science/pith/IEWBA3OEGRWZLXTMDPVAPATG3Y/action/citation_signature","submit_replication":"https://pith.science/pith/IEWBA3OEGRWZLXTMDPVAPATG3Y/action/replication_record"}},"created_at":"2026-07-05T11:53:48.842676+00:00","updated_at":"2026-07-05T11:53:48.842676+00:00"}