{"as_of":"2026-08-08T02:31:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a1e1ead8ef6e06d250a7e382a8f9fa238719ed0759108c20949f4f07a477f1d4","coverage":[{"denominator":61,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":61,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T17:32:34.292050Z","state":"measured"},{"denominator":72,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":72,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":11,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":11,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T20:21:21.808023Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-07T12:53:50.345403Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.10535","snapshot_observed_at":"2026-08-04T20:21:21.808023Z","title":"Jiang, S.; Huang, Z.; Luo, X.; and Sun, Y","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.08682","last_updated":"2025-09-10T15:22:00Z","snapshot_observed_at":"2026-08-07T12:17:08.646686Z","submitted_at":"2025-09-10T15:22:00Z","title":"Automatic Failure Attribution and Critical Step Prediction Method for Multi-Agent Systems Based on Causal Inference","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:21.808023Z"},"links":{"cited_paper":"/paper/2507.10535","citing_paper":"/paper/2509.08682"},"observation_digest":"sha256:edcf2aa7bb9b2e996e91f72670cfe3280829e2178c795d788f12fac7a3c34f9f","observation_id":"081fa9fd-8790-4a89-a31e-c299e3a060b9","resolution":{"observed_at":"2026-08-04T20:21:21.808023Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"cited_work":{"arxiv_id":"2507.10535","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.10535","snapshot_observed_at":"2026-07-07T12:53:50.345403Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-judge for coding tasks","venue":"cs.CL","work_id":"7e37bba3-99fb-4b09-af65-f199ccc4101e","year":2025},"citing_paper":{"arxiv_id":"2509.09936","last_updated":"2025-09-12T02:53:57Z","snapshot_observed_at":"2026-07-06T22:29:06.119014Z","submitted_at":"2025-09-12T02:53:57Z","title":"SciML Agents: Write the Solver, Not the Solution","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-18T17:57:51.444493Z"},"links":{"cited_paper":"/paper/2507.10535","citing_paper":"/paper/2509.09936"},"observation_digest":"sha256:18a7cf92a4f91596f99628f1f674c92026f70d34ffb9cfabe43ac0f686cd0956","observation_id":"0b3b896f-db59-46f6-85d3-2005ee87ec9d","resolution":{"observed_at":"2026-05-18T18:01:43.843517Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"cited_work":{"arxiv_id":"2507.10535","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.10535","snapshot_observed_at":"2026-07-07T12:53:50.345403Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-judge for coding tasks","venue":"cs.CL","work_id":"7e37bba3-99fb-4b09-af65-f199ccc4101e","year":2025},"citing_paper":{"arxiv_id":"2604.16790","last_updated":"2026-04-18T02:35:05Z","snapshot_observed_at":"2026-07-06T23:04:01.558812Z","submitted_at":"2026-04-18T02:35:05Z","title":"Bias in the Loop: Auditing LLM-as-a-Judge for Software Engineering","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-10T07:29:03.994957Z"},"links":{"cited_paper":"/paper/2507.10535","citing_paper":"/paper/2604.16790"},"observation_digest":"sha256:8a368591547569469bcdde116bdb8e9fb2791f73fca7c6dd3c4a2be202c605a7","observation_id":"9c5510e2-213a-4a3a-98d2-e90d1936e3c6","resolution":{"observed_at":"2026-05-10T07:32:00.313227Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"cited_work":{"arxiv_id":"2507.10535","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.10535","snapshot_observed_at":"2026-07-07T12:53:50.345403Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-judge for coding tasks","venue":"cs.CL","work_id":"7e37bba3-99fb-4b09-af65-f199ccc4101e","year":2025},"citing_paper":{"arxiv_id":"2604.27727","last_updated":"2026-04-30T11:20:22Z","snapshot_observed_at":"2026-07-06T23:13:11.623453Z","submitted_at":"2026-04-30T11:20:22Z","title":"LLM-as-a-Judge for Human-AI Co-Creation: A Reliability-Aware Evaluation Framework for Coding","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-07T08:39:55.256518Z"},"links":{"cited_paper":"/paper/2507.10535","citing_paper":"/paper/2604.27727"},"observation_digest":"sha256:3b6a81cfee11356c852227203fd3a81ce9c523cc7a4aae88a06cf12221a36dd5","observation_id":"f0897a24-f187-4a50-ba93-1a1b3bbfa912","resolution":{"observed_at":"2026-05-12T09:56:28.093044Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"cited_work":{"arxiv_id":"2507.10535","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.10535","snapshot_observed_at":"2026-07-07T12:53:50.345403Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-judge for coding tasks","venue":"cs.CL","work_id":"7e37bba3-99fb-4b09-af65-f199ccc4101e","year":2025},"citing_paper":{"arxiv_id":"2605.01474","last_updated":"2026-05-02T14:44:49Z","snapshot_observed_at":"2026-07-06T23:14:43.034336Z","submitted_at":"2026-05-02T14:44:49Z","title":"ReMedi: Reasoner for Medical Clinical Prediction","version":1},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-05-09T14:20:29.672994Z"},"links":{"cited_paper":"/paper/2507.10535","citing_paper":"/paper/2605.01474"},"observation_digest":"sha256:ff53cf816dd8ab45664623d36166f35e5717cc6b93fa38d906d913d3189d259a","observation_id":"5095cda1-e4e3-45de-acf0-126397d21c30","resolution":{"observed_at":"2026-05-11T17:01:05.917444Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"cited_work":{"arxiv_id":"2507.10535","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.10535","snapshot_observed_at":"2026-07-07T12:53:50.345403Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-judge for coding tasks","venue":"cs.CL","work_id":"7e37bba3-99fb-4b09-af65-f199ccc4101e","year":2025},"citing_paper":{"arxiv_id":"2605.02906","last_updated":"2026-07-31T03:14:41Z","snapshot_observed_at":"2026-08-05T23:10:40.522491Z","submitted_at":"2026-04-06T02:40:18Z","title":"OpsLLM: Construction of Large Language Model for Software Operations with Multi-stage Learning","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-10T20:07:07.548384Z"},"links":{"cited_paper":"/paper/2507.10535","citing_paper":"/paper/2605.02906"},"observation_digest":"sha256:0df9385c6fa69ac4cfcbd1566c57af83b08847571ad6ac87a2365b834134d236","observation_id":"4ab9dd30-515b-4947-a9b1-41a95316c4bc","resolution":{"observed_at":"2026-05-10T22:15:48.632378Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"cited_work":{"arxiv_id":"2507.10535","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.10535","snapshot_observed_at":"2026-07-07T12:53:50.345403Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-judge for coding tasks","venue":"cs.CL","work_id":"7e37bba3-99fb-4b09-af65-f199ccc4101e","year":2025},"citing_paper":{"arxiv_id":"2605.02906","last_updated":"2026-07-31T03:14:41Z","snapshot_observed_at":"2026-08-05T23:10:40.522491Z","submitted_at":"2026-04-06T02:40:18Z","title":"OpsLLM: Construction of Large Language Model for Software Operations with Multi-stage Learning","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-13T06:25:16.650306Z"},"links":{"cited_paper":"/paper/2507.10535","citing_paper":"/paper/2605.02906"},"observation_digest":"sha256:dfddbcf66ff01a9e9606d461129222a2403593b2453e2ce4a04f175b0ec636c7","observation_id":"5e398c30-299a-4459-83a9-72ed8fe09a01","resolution":{"observed_at":"2026-05-13T06:27:24.609554Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.10535","snapshot_observed_at":"2026-08-03T02:30:32.418305Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.02906","last_updated":"2026-07-31T03:14:41Z","snapshot_observed_at":"2026-08-05T23:10:40.522491Z","submitted_at":"2026-04-06T02:40:18Z","title":"OpsLLM: Construction of Large Language Model for Software Operations with Multi-stage Learning","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-03T02:30:32.418305Z"},"links":{"cited_paper":"/paper/2507.10535","citing_paper":"/paper/2605.02906"},"observation_digest":"sha256:c5b1f72b59d2fd964f021a7ea75105a81aab36d206bbaed1c1675cd0a045ebc8","observation_id":"53877acc-d601-4396-a639-446a0da43499","resolution":{"observed_at":"2026-08-03T02:30:32.418305Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"cited_work":{"arxiv_id":"2507.10535","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.10535","snapshot_observed_at":"2026-07-07T12:53:50.345403Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-judge for coding tasks","venue":"cs.CL","work_id":"7e37bba3-99fb-4b09-af65-f199ccc4101e","year":2025},"citing_paper":{"arxiv_id":"2605.13139","last_updated":"2026-05-13T08:05:16Z","snapshot_observed_at":"2026-08-08T01:44:09.576347Z","submitted_at":"2026-05-13T08:05:16Z","title":"SWE-Cycle: Benchmarking Code Agents across the Complete Issue Resolution Cycle","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-14T18:34:39.997353Z"},"links":{"cited_paper":"/paper/2507.10535","citing_paper":"/paper/2605.13139"},"observation_digest":"sha256:b6c5bddea3c665055f57bcbe7e97c281841cf7bb3ecb1d527ac1bd9a9c60c28c","observation_id":"7e8a0566-9712-4552-86ea-45b3bb2a9231","resolution":{"observed_at":"2026-05-14T18:37:35.746653Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"cited_work":{"arxiv_id":"2507.10535","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.10535","snapshot_observed_at":"2026-07-07T12:53:50.345403Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-judge for coding tasks","venue":"cs.CL","work_id":"7e37bba3-99fb-4b09-af65-f199ccc4101e","year":2025},"citing_paper":{"arxiv_id":"2607.05391","last_updated":"2026-07-07T17:26:37Z","snapshot_observed_at":"2026-08-06T03:00:58.001418Z","submitted_at":"2026-07-06T17:59:35Z","title":"LLM-as-a-Verifier: A General-Purpose Verification Framework","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-07-07T12:47:29.552283Z"},"links":{"cited_paper":"/paper/2507.10535","citing_paper":"/paper/2607.05391"},"observation_digest":"sha256:bd49b3796932d3394c1543d4e644e875db22ebb3683c8200c2a659f39370ef0b","observation_id":"c47fe0ad-dcbc-4e60-ba75-5ed87aa6e5fe","resolution":{"observed_at":"2026-07-07T12:53:50.347229Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.10535","snapshot_observed_at":"2026-07-11T07:02:51.850836Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.05391","last_updated":"2026-07-07T17:26:37Z","snapshot_observed_at":"2026-08-06T03:00:58.001418Z","submitted_at":"2026-07-06T17:59:35Z","title":"LLM-as-a-Verifier: A General-Purpose Verification Framework","version":2},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-07-11T07:02:51.850836Z"},"links":{"cited_paper":"/paper/2507.10535","citing_paper":"/paper/2607.05391"},"observation_digest":"sha256:e95eb2626d632659163da9ecf42aee1ef4a24f5ee6f9b0bb372d12ae5b9c94da","observation_id":"b7a9f655-9f30-4eb0-99d8-c68ab6d5b659","resolution":{"observed_at":"2026-07-11T07:02:51.850836Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2507.10535/citation-record","integrity":"/paper/2507.10535/integrity","json":"/paper/2507.10535/citation-record.json","paper":"/paper/2507.10535"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2412.08905","last_updated":"2024-12-12T03:37:41Z","snapshot_observed_at":"2026-08-05T04:04:21.846023Z","submitted_at":"2024-12-12T03:37:41Z","title":"Phi-4 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.08905","snapshot_observed_at":"2026-08-06T17:32:29.662597Z","title":"Phi-4 technical report","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:29.662597Z"},"links":{"cited_paper":"/paper/2412.08905","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:f23c8daa31f59782522c176516eebba2d1fafb473b75a7918406a22a99492000","observation_id":"0c9225aa-51d4-4752-85d4-1bcb44500da7","resolution":{"observed_at":"2026-08-06T17:32:29.662597Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-06T17:32:29.722527Z","title":"Gpt-4 technical report","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:29.722527Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:d3444c9029f2ffd455d45117832771b67cdd59b900d228a774b20fa99d340b5e","observation_id":"9c8434c2-b472-434c-b1f1-b5318c68cdd3","resolution":{"observed_at":"2026-08-06T17:32:29.722527Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:42.676748Z","title":"Automated unit test improvement using large language models at meta","venue":null,"work_id":"aa33e1dd-bab4-442c-9fb3-d0a1866555bc","year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:29.910326Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:8cb56740dd470edab301012167f245fb83f3643a408c179efd94afd37c24e1a4","observation_id":"e1e60d71-f17a-48b4-a3ee-b24feed19bc5","resolution":{"observed_at":"2026-08-06T17:32:42.690068Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:42.489668Z","title":"Claude 3.7","venue":null,"work_id":"4014125b-01cd-4341-ade8-195666915441","year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:30.002094Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:478688e07df60a3320e4c7dbc225d069b1b3ffb7a7898395d739b79348fc1782","observation_id":"62df6330-63b9-4093-95d4-df8bb968ef8c","resolution":{"observed_at":"2026-08-06T17:32:42.581478Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:42.290037Z","title":"Claude 4","venue":null,"work_id":"0c2d82ad-6031-4de2-a40d-c63f141ac9b1","year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:30.072314Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:03d418a65a17172f97a03c6844bd943fac0cd95f1f3e6d80e35fd54dae29f508","observation_id":"9341a860-3a15-4d0d-b84c-f39b31c1ff6a","resolution":{"observed_at":"2026-08-06T17:32:42.378283Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2108.07732","last_updated":"2021-08-16T03:57:30Z","snapshot_observed_at":"2026-08-02T19:23:53.535075Z","submitted_at":"2021-08-16T03:57:30Z","title":"Program Synthesis with Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2108.07732","snapshot_observed_at":"2026-08-06T17:32:30.162463Z","title":"Program synthesis with large language models","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:30.162463Z"},"links":{"cited_paper":"/paper/2108.07732","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:feb82513f7c966654d4be36bc8fa2f1c4c1bc1924298d80d43a05d3ced7226e8","observation_id":"3bd8cac9-39a9-4a64-95fb-fc03c8fa2508","resolution":{"observed_at":"2026-08-06T17:32:30.162463Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:30.256357Z","title":"Codet: Code generation with generated tests","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:30.256357Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:a87934fc1a640aac628876a2812b13ff812bbaacc066983e26904cf09fc7e938","observation_id":"0e4ec585-5f9d-41e7-979f-cb0bb4d58296","resolution":{"observed_at":"2026-08-06T17:32:30.256357Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2107.03374","last_updated":"2021-07-14T17:16:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-07-07T17:41:24Z","title":"Evaluating Large Language Models Trained on Code","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2107.03374","snapshot_observed_at":"2026-08-06T17:32:30.358202Z","title":"Evaluating large language models trained on code","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:30.358202Z"},"links":{"cited_paper":"/paper/2107.03374","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:a8b8e396bede0e346bc68dc57c0c9a6cc272e913fc9b194afb9bfa02e19ae3df","observation_id":"45ae6b93-0aed-4333-8e5a-2843f38df120","resolution":{"observed_at":"2026-08-06T17:32:30.358202Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:42.158114Z","title":"Teaching large language models to self-debug","venue":null,"work_id":"196967b6-0711-4769-9583-9027c1e19496","year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:30.467485Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:e2ed0df0640c6151970039603d41065bcd158892c525b8e1dfff91ec68b02379","observation_id":"d409d1fb-c907-4c60-987b-00016a2eb356","resolution":{"observed_at":"2026-08-06T17:32:42.179788Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:30.562154Z","title":"Rm-r1: Reward modeling as reasoning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:30.562154Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:edff1a525da6984f547be84116fd6092430eb5585ea3d29d82db55ed3b1ec5bd","observation_id":"8f725fbe-d2e9-48e0-9fb4-00a9eb87b637","resolution":{"observed_at":"2026-08-06T17:32:30.562154Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.16400","last_updated":"2025-06-05T17:59:12Z","snapshot_observed_at":"2026-08-08T01:11:54.830004Z","submitted_at":"2025-05-22T08:50:47Z","title":"AceReason-Nemotron: Advancing Math and Code Reasoning through Reinforcement Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.16400","snapshot_observed_at":"2026-08-06T17:32:30.634578Z","title":"Acereason-nemotron: Advancing math and code reasoning through reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:30.634578Z"},"links":{"cited_paper":"/paper/2505.16400","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:64fe8d990e0fe6b0d86af31914533cb8230e6790cc3457a296f5925f4e887558","observation_id":"6712258d-8282-40f6-ad1f-31d0bc269b16","resolution":{"observed_at":"2026-08-06T17:32:30.634578Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:41.985542Z","title":null,"venue":null,"work_id":"69a6f000-ad38-4cf0-8c2a-da297d44ec86","year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:30.697316Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:d83fe899152f29ea7f553f794c5eae5c014ed43b6ec1781d074a9359d1ae7f0d","observation_id":"695a6cdc-7378-42b7-bcb5-426c16d294c5","resolution":{"observed_at":"2026-08-06T17:32:42.080154Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-06T17:32:30.805802Z","title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:30.805802Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:1abf1d1596e6184838b68837a7c47771fe1e022861eaa94e8f3907d10837afb1","observation_id":"410a27dd-5851-4e7c-8697-13ed5db91dc1","resolution":{"observed_at":"2026-08-06T17:32:30.805802Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.06782","last_updated":"2024-06-02T16:16:49Z","snapshot_observed_at":"2026-08-07T16:46:33.307475Z","submitted_at":"2023-08-13T14:35:50Z","title":"PentestGPT: An LLM-empowered Automatic Penetration Testing Tool","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.06782","snapshot_observed_at":"2026-08-06T17:32:30.900133Z","title":"Pentestgpt: An llm-empowered automatic penetration testing tool","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:30.900133Z"},"links":{"cited_paper":"/paper/2308.06782","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:c4ec9fa6cc8b44bf60554e066de30dd544a6dd8b03297b34e996443d82f75d81","observation_id":"3b843fee-cf32-4efb-88a2-681611e71858","resolution":{"observed_at":"2026-08-06T17:32:30.900133Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14723","last_updated":"2025-02-03T18:57:05Z","snapshot_observed_at":"2026-07-06T20:25:45.259430Z","submitted_at":"2025-01-24T18:58:40Z","title":"CodeMonkeys: Scaling Test-Time Compute for Software Engineering","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.14723","snapshot_observed_at":"2026-08-06T17:32:30.995056Z","title":"Codemonkeys: Scaling test-time compute for software engineering","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:30.995056Z"},"links":{"cited_paper":"/paper/2501.14723","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:20883b37e4f33b7046de409b9055a2e70e9941801ee6c9c1f4f558ca73543f7b","observation_id":"a506cdaf-03d9-4bbf-8546-4787acd178c7","resolution":{"observed_at":"2026-08-06T17:32:30.995056Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13820","last_updated":"2025-07-30T14:58:42Z","snapshot_observed_at":"2026-08-07T18:04:41.000672Z","submitted_at":"2025-02-19T15:32:11Z","title":"Scoring Verifiers: Evaluating Synthetic Verification for Code and Reasoning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.13820","snapshot_observed_at":"2026-08-06T17:32:31.064570Z","title":"Scoring verifiers: Eval- uating synthetic verification for code and reasoning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:31.064570Z"},"links":{"cited_paper":"/paper/2502.13820","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:c85c5885bcb35fbd26608b86f3ea8ce6baf548804976597841896b2e12778a22","observation_id":"e5c1bca7-a3d5-4dbe-8616-345224c88e81","resolution":{"observed_at":"2026-08-06T17:32:31.064570Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:41.759917Z","title":"Gonzalez, and Ion Stoica","venue":null,"work_id":"b963f2e9-ed30-4b5e-a6ba-5204cc6810fe","year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:31.155235Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:59e03854eb8073d5b7aefceaa22c857c1aac97ac53bf166e01d46f639b36dbf9","observation_id":"95dd83da-1adb-44f4-ad80-4d83218f093d","resolution":{"observed_at":"2026-08-06T17:32:41.850357Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-06T17:32:31.253381Z","title":"The llama 3 herd of models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:31.253381Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:60d266ab887eb9a9d0b7497b8df5b09658c6aa63b90efeff8cd3c51deceaaf4e","observation_id":"d941a680-1c4d-4da4-924e-50dd858da513","resolution":{"observed_at":"2026-08-06T17:32:31.253381Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15594","last_updated":"2025-10-19T10:32:43Z","snapshot_observed_at":"2026-08-02T10:23:50.881300Z","submitted_at":"2024-11-23T16:03:35Z","title":"A Survey on LLM-as-a-Judge","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15594","snapshot_observed_at":"2026-08-06T17:32:31.353188Z","title":"A survey on llm-as-a-judge","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:31.353188Z"},"links":{"cited_paper":"/paper/2411.15594","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:f4bd852ca9519f9fbaa0f85332dde7aabf83db757f2b026d2cd3034bc5ff31a1","observation_id":"316e3584-dd13-49d2-bfa7-fb34d882d5d8","resolution":{"observed_at":"2026-08-06T17:32:31.353188Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.02246","last_updated":"2025-03-04T03:48:23Z","snapshot_observed_at":"2026-08-07T17:31:39.128866Z","submitted_at":"2025-03-04T03:48:23Z","title":"From Code to Courtroom: LLMs as the New Software Judges","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.02246","snapshot_observed_at":"2026-08-06T17:32:31.444375Z","title":"From code to courtroom: Llms as the new software judges","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:31.444375Z"},"links":{"cited_paper":"/paper/2503.02246","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:e24627e414f8c3cbee0c4b7ffc475e40bfc205d877572a78b0d72bc6d2ff5c81","observation_id":"6c19e092-2e61-4550-8117-b1a7ec4a4f62","resolution":{"observed_at":"2026-08-06T17:32:31.444375Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:41.591732Z","title":"An empirical study on fine-tuning large language models of code for automated program repair","venue":null,"work_id":"81fe7dc1-7695-444c-a920-c74f1d0819d7","year":2023},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:31.535547Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:798a3b0a88b96f6207e97e937a47c0700a50050403acd5949dbdd499afd6821d","observation_id":"3647b23d-a7eb-49a6-befb-0dbbc17e4742","resolution":{"observed_at":"2026-08-06T17:32:41.666376Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:41.243654Z","title":"Livecodebench: Holistic and contamination free evaluation of large language models for code","venue":null,"work_id":"3a20a727-33d5-4194-b5c6-924ac2dea58a","year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:31.724734Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:acdc3fda88aaf7cad69a81ed6f463bae63f7315c12a4a67ceba64a91481dfb2d","observation_id":"eaa332d2-4a91-4484-ba4b-004a39388cfc","resolution":{"observed_at":"2026-08-06T17:32:41.298959Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:41.067413Z","title":"Self-planning code generation with large language models","venue":null,"work_id":"4f043466-0623-449a-9c8f-7154a68b5c06","year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:31.808480Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:377beebfcf0c5156144b0804d5ad766e9fd31f604c87773f1aaa663dfcd9937b","observation_id":"578fa9a7-ae6c-47e5-849d-3318038fe9c7","resolution":{"observed_at":"2026-08-06T17:32:41.116337Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:40.870737Z","title":"Critiquellm: Towards an informative critique generation model for evaluation of large language model generation","venue":null,"work_id":"1ec4df92-a32f-4d41-8c3f-e48f81b7252c","year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:31.884269Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:6790b8e15dd9680c3b345118a6ae30bac051340da73c1a9932c7414bdff942ce","observation_id":"ef81ede8-6741-4d9a-be3c-9706e88a8122","resolution":{"observed_at":"2026-08-06T17:32:40.948898Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:40.643469Z","title":"Welleck, Graham Neubig, Moontae Lee, Kyungjae Lee, and Minjoon Seo","venue":null,"work_id":"239bf8f2-ca26-4f06-b9a7-d54f95e59221","year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:31.947036Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:2a401d6938ce6a36f543c2273090925cbee47d6eaab226c9a409a7c59b861fb0","observation_id":"a3d91791-856c-4d30-bea3-b492ecddfe3b","resolution":{"observed_at":"2026-08-06T17:32:40.717471Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:40.531484Z","title":"Overfitting in semantics-based automated program repair","venue":null,"work_id":"d1c444dd-8b90-4e14-a16c-45bf476e32eb","year":2018},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:31.986536Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:da80093354454591e2513dd33ad5c9056c6271a4a141645efe58d41c1c4338d2","observation_id":"b74c1185-6ad2-44c5-af6c-4095df969310","resolution":{"observed_at":"2026-08-06T17:32:40.627958Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:40.309690Z","title":"Generative judge for evaluating alignment","venue":null,"work_id":"40278fa4-b41e-42c6-b573-bf4bfb5fb373","year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:32.050286Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:4e61f908d4a4f3020c43d0cfc8458964c0fb81778b917c8f67d6ecc5b6efcd1c","observation_id":"2ddd0d21-3a93-4d42-b139-cdbb69f0486d","resolution":{"observed_at":"2026-08-06T17:32:40.377848Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:40.121493Z","title":"Competition-level code generation with alphacode","venue":null,"work_id":"5cce0596-47d9-42cb-b554-b3e5ac88aa7b","year":2022},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:32.125628Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:3d857b3f6b3c8a28221b3354077cfd91d589c82f4caab71f6478a81424e6dd3d","observation_id":"73900c1d-eef2-4ca9-9e17-3bbe1bb4b175","resolution":{"observed_at":"2026-08-06T17:32:40.177026Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:39.911338Z","title":"Llms for relational reasoning: How far are we? In Proceedings of the 1st International Workshop on Large Language Models for Code, pages 119–126, 2024","venue":null,"work_id":"4fa89a2d-a66e-4def-899b-c63ee82fff16","year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:32.215870Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:ec3b5555d0bf5320239bb23493c937d310f27c18e39138909d5b0c8f68f58dea","observation_id":"e923a080-abd6-415f-a6fd-116562346624","resolution":{"observed_at":"2026-08-06T17:32:39.982049Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:39.703264Z","title":"RM-bench: Benchmarking reward models of language models with subtlety and style","venue":null,"work_id":"cc45c4d8-76c7-45ab-8320-5e31c0d0c31c","year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:32.300774Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:621a8a44d94bdd18cac030813baad1b9786c672fb7c6fdf6d26a73cc2928c9fc","observation_id":"82359399-10d2-4abb-8e8b-46f46f96a630","resolution":{"observed_at":"2026-08-06T17:32:39.801044Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:39.434975Z","title":"Deepcoder: A fully open-source 14b coder at o3-mini level","venue":null,"work_id":"0848e680-27ab-4e16-a490-6639a313da12","year":null},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:32.397991Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:29842e4621ed6af74409363e1944c35842a5cfb83869afb9f262740ae9f82dfe","observation_id":"631f1674-bce1-43d5-83b6-4f0f63460f6d","resolution":{"observed_at":"2026-08-06T17:32:39.578462Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00215","last_updated":"2024-06-28T19:53:17Z","snapshot_observed_at":"2026-07-06T18:38:46.314431Z","submitted_at":"2024-06-28T19:53:17Z","title":"LLM Critics Help Catch LLM Bugs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.00215","snapshot_observed_at":"2026-08-06T17:32:32.445876Z","title":"Llm critics help catch llm bugs","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:32.445876Z"},"links":{"cited_paper":"/paper/2407.00215","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:b58d5a3d07f903e042cc46f3acd6cbb80771ca462e27f71a2154c253928c9b66","observation_id":"6868fb9f-f436-4f7d-959a-d7d92e1ba5c1","resolution":{"observed_at":"2026-08-06T17:32:32.445876Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:39.251400Z","title":"Swt-bench: Testing and validating real- world bug-fixes with code agents","venue":null,"work_id":"a5f1d2bd-1108-4bb8-a111-104ec21d89ca","year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:32.554314Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:525535d416f4755835ffc5c1f0dc607a6e2d24f80dbed67722ab61f08f6dc022","observation_id":"829352c5-1163-4e92-8643-563012c9b409","resolution":{"observed_at":"2026-08-06T17:32:39.338250Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:39.066987Z","title":"Wainwright, Pamela Mishkin, Chong Zhang, Sandhini Agarwal, Katarina Slama, Alex Ray, John Schulman, Jacob Hilton, Fraser Kelton, Luke E","venue":null,"work_id":"cd754242-3fee-4abb-b7d0-8bf7613a7c55","year":2022},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:32.619413Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:fbcf9d5af3ad7634605ef6a0a55193d678d0f1006997138228ab17852302150f","observation_id":"472ed59f-d8a6-4962-bb05-5172541db030","resolution":{"observed_at":"2026-08-06T17:32:39.158021Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:32.748374Z","title":"M-prometheus: A suite of open multilingual llm judges","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:32.748374Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:ea384cf52f086df39ce1f02a9be4721925e4273aebeb79035501a3ee469fe82d","observation_id":"5ebb1ec5-b98a-4307-8344-71d3e751d1ce","resolution":{"observed_at":"2026-08-06T17:32:32.748374Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2009.10297","last_updated":"2020-09-27T04:07:11Z","snapshot_observed_at":"2026-08-01T07:33:26.380394Z","submitted_at":"2020-09-22T03:10:49Z","title":"CodeBLEU: a Method for Automatic Evaluation of Code Synthesis","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.10297","snapshot_observed_at":"2026-08-06T17:32:32.851096Z","title":"Codebleu: a method for automatic evaluation of code synthesis","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:32.851096Z"},"links":{"cited_paper":"/paper/2009.10297","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:e534c0d15c0603de92006359d6889449806ad044b08605e22ceaebda5e6bbbcb","observation_id":"37fc0660-7375-480d-81b2-a5f9e0a455bc","resolution":{"observed_at":"2026-08-06T17:32:32.851096Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:38.700270Z","title":"Skywork critic model series","venue":null,"work_id":"066f62eb-2e82-44f0-ba80-7358ae83470c","year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:32.948957Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:2c00dab29feb941678604074e0f6ce4c5df0493c2e9448221965295b4743ad08","observation_id":"021bad87-7e6c-4c79-9237-9370d53118ce","resolution":{"observed_at":"2026-08-06T17:32:38.843015Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03314","last_updated":"2024-08-06T17:35:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:35:05Z","title":"Scaling LLM Test-Time Compute Optimally can be More Effective than Scaling Model Parameters","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03314","snapshot_observed_at":"2026-08-06T17:32:33.016659Z","title":"Scaling llm test-time compute optimally can be more effective than scaling model parameters","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.016659Z"},"links":{"cited_paper":"/paper/2408.03314","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:657a7337895c86a07c765a40491a313eb33aad055e08168fa4b3a28d79ed8fb7","observation_id":"0501d0ec-bd1a-40fc-9468-b84909a2b173","resolution":{"observed_at":"2026-08-06T17:32:33.016659Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:38.543059Z","title":"Judgebench: A benchmark for evaluating LLM-based judges","venue":null,"work_id":"3cf8cb53-f1d3-4471-90db-72a01faf713c","year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.074682Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:f4841fa1db8dce2ac14993f23d4c070eed2adfbb6c0f2c06d4e104acc914cbf7","observation_id":"8511f9d0-f646-4aae-9251-be4967b4aa68","resolution":{"observed_at":"2026-08-06T17:32:38.624158Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:38.359660Z","title":"Code repair with LLMs gives an exploration-exploitation tradeoff","venue":null,"work_id":"f66f49ec-7ac0-4316-8440-047f960d6103","year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.171459Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:e67e0416f62a517138ff5d4ffb4715f977af63223cd32c65174ce20c1a3be343","observation_id":"c7f9eda7-12be-4143-a706-61e625c8f76d","resolution":{"observed_at":"2026-08-06T17:32:38.439815Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:33.228386Z","title":"Qwq-32b: Embracing the power of reinforcement learning, March 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.228386Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:c253658b46ab6af0379decc06ed2304a2ef7e6702172eedf7ce7d48eb1d2999e","observation_id":"c5763663-0db8-4bcc-9788-2daf9c7148fb","resolution":{"observed_at":"2026-08-06T17:32:33.228386Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:38.043255Z","title":"Can llms replace human evaluators? an empirical study of llm-as-a-judge in software engineering","venue":null,"work_id":"f2de2f64-a0ba-4e85-aa6a-884189e4d751","year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.302701Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:e9532fe4a8fd3736209f3d6a062d57b99c7f4a75a577c2d05bcf079c892651ee","observation_id":"140f308e-9e28-4b3a-b437-16554802428b","resolution":{"observed_at":"2026-08-06T17:32:38.150125Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.02666","last_updated":"2024-08-08T17:09:58Z","snapshot_observed_at":"2026-08-06T10:06:21.140448Z","submitted_at":"2024-08-05T17:57:02Z","title":"Self-Taught Evaluators","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.02666","snapshot_observed_at":"2026-08-06T17:32:33.368009Z","title":"Self-taught evaluators","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.368009Z"},"links":{"cited_paper":"/paper/2408.02666","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:d033131b5e06a33d1f772fd69bd01519836a4c05dc7ec13f033f81c45c098c64","observation_id":"ac0be8b9-04c1-4ea2-a14f-64a21bb03f77","resolution":{"observed_at":"2026-08-06T17:32:33.368009Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:37.873991Z","title":"PandaLM: An automatic evaluation benchmark for LLM instruction tuning optimization","venue":null,"work_id":"229d215f-e628-45fc-9c3e-6b62ed9ef769","year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.423579Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:d32aadaf7a8fc9368dda196677620b3e5178c1ddfd50f8d239203f65292c693e","observation_id":"96dfe862-aef5-47d9-8556-c1c967cc724d","resolution":{"observed_at":"2026-08-06T17:32:37.938874Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:37.630252Z","title":"Chi, Tatsunori Hashimoto, O","venue":null,"work_id":"7f1c7d3e-0041-4cf3-a23e-d1a76fe37a9a","year":2022},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.480092Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:d270c4c57966d39343e05d1eddafef252b4140ba36df333e7e35c51e42dbf150","observation_id":"d02dbe15-a1e3-44b9-8637-b54352e5b1d8","resolution":{"observed_at":"2026-08-06T17:32:37.753167Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:37.341324Z","title":"Weyssow, Aton Kamanda, Xin Zhou, and H","venue":null,"work_id":"e348e673-976e-4eb9-b23c-6fa989346666","year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.542361Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:25cd9754bf7a5df2a84dae5ffe272f53c56f9fae0b39794f6f5cff2a86cbce60","observation_id":"88c200d2-e87a-4020-99f4-0866db8a631e","resolution":{"observed_at":"2026-08-06T17:32:37.467984Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.17564","last_updated":"2023-12-21T06:21:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-30T17:30:36Z","title":"BloombergGPT: A Large Language Model for Finance","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.17564","snapshot_observed_at":"2026-08-06T17:32:33.592955Z","title":"Bloomberggpt: A large language model for finance","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.592955Z"},"links":{"cited_paper":"/paper/2303.17564","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:3d747ea1cd9bab81d528104d16c69431b64aa8fe7e9361c727b2d4a01abf8f28","observation_id":"315537e2-45d4-4d6b-8b39-5b45057ab3ba","resolution":{"observed_at":"2026-08-06T17:32:33.592955Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-08-06T17:32:33.649376Z","title":"Qwen3 technical report","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.649376Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:bf5314df95384a3c8275cc195794b03813187745ad0085bae785e477732c389c","observation_id":"e1b57fa7-d8c3-4db2-9623-1166c2b32fa1","resolution":{"observed_at":"2026-08-06T17:32:33.649376Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-06T17:32:33.704634Z","title":"Qwen2.5 technical report","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.704634Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:b3771c28eb33da801f04f5d123773a8dec8235dce669537251c01e638b74ea5f","observation_id":"c8630a18-a370-41d3-b810-fce5a54e322b","resolution":{"observed_at":"2026-08-06T17:32:33.704634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19502","last_updated":"2025-05-26T04:29:14Z","snapshot_observed_at":"2026-08-08T01:24:24.651088Z","submitted_at":"2025-05-26T04:29:14Z","title":"CODE-DITING: A Reasoning-Based Metric for Functional Alignment in Code Evaluation","version":1},"cited_work":{"arxiv_id":"2505.19502","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.19502","snapshot_observed_at":"2026-08-06T17:32:34.591005Z","title":"CODE-DITING: A Reasoning-Based Metric for Functional Alignment in Code Evaluation","venue":"cs.SE","work_id":"761a9802-1014-408b-ae6d-a7bd93131e61","year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.789468Z"},"links":{"cited_paper":"/paper/2505.19502","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:98f66d58b6cdc52b8f29d12993f98de08081421da3caf4e39ef6beafa41e9574","observation_id":"3578691e-6982-4de8-8a26-b88ee4ce7fbf","resolution":{"observed_at":"2026-08-06T17:32:34.645100Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:33.841936Z","title":"Fingpt: Open-source financial large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.841936Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:9d62465e681b1f95fb61f68e0b5323358ef6c26ee4ea2f1fe400ed509abfb10a","observation_id":"03322681-7dee-498c-a3c3-3894d5dd2a95","resolution":{"observed_at":"2026-08-06T17:32:33.841936Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:37.111759Z","title":"Demystifying long chain-of- thought reasoning in llms, 2025","venue":null,"work_id":"db0d7318-96f4-4fe6-aff7-02b55ae1fc7c","year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.881950Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:8dba0d466ec3d45812694bbcf958e622132d91e9e158135d0e87a7b1d98dabb8","observation_id":"c3182861-bd31-4aff-8784-a6e4d46520d5","resolution":{"observed_at":"2026-08-06T17:32:37.208972Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01718","last_updated":"2025-05-24T04:36:48Z","snapshot_observed_at":"2026-07-06T20:30:31.588538Z","submitted_at":"2025-02-03T18:46:04Z","title":"ACECODER: Acing Coder RL via Automated Test-Case Synthesis","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.01718","snapshot_observed_at":"2026-08-06T17:32:33.941396Z","title":"Acecoder: Acing coder rl via automated test-case synthesis","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.941396Z"},"links":{"cited_paper":"/paper/2502.01718","citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:21961b37392276d082e6e73d287697d9a106f2e8d809a2ff8c4b0d15844bfba7","observation_id":"de824f34-39f5-4063-9f92-02c42a138841","resolution":{"observed_at":"2026-08-06T17:32:33.941396Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:36.818079Z","title":"Codecriticbench: A holistic code critique benchmark for large language models, 2025","venue":null,"work_id":"28831811-ff04-4c8e-80f9-bda739e00b3c","year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:33.991009Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:5aef32201d6bfc0ba3bf2b2c076db6a71ca550a6d6284bc8f990d84bffe88101","observation_id":"14f188bb-f344-4d63-b28a-ffffff9a306f","resolution":{"observed_at":"2026-08-06T17:32:36.936289Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:36.528012Z","title":null,"venue":null,"work_id":"6af7d204-bcbe-49e9-9875-6e4c941889e8","year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:34.024035Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:fd922d83f62496e16cea70b1426a7853fcebb53dc2505d488dc46612817ebe10","observation_id":"e11e4bc7-896e-4b54-b34e-7ea4f720d65c","resolution":{"observed_at":"2026-08-06T17:32:36.639897Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:34.062676Z","title":"Judging llm-as-a-judge with mt-bench and chatbot arena","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:34.062676Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:3a0560c8eb5b857f89c502ed3493d72b25595a50ded2c39924b434a12c2ec73e","observation_id":"9f89e6cd-5739-484b-be82-2f5872ce34bf","resolution":{"observed_at":"2026-08-06T17:32:34.062676Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:36.241147Z","title":"RMB: Compre- hensively benchmarking reward models in LLM alignment","venue":null,"work_id":"5353d0c8-2cd0-489e-abe8-95947fd0ad23","year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:34.116865Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:668049c42b8deefaf4833a16ab58c1c7b2e57afc7547fbe14d0dbd351af05f07","observation_id":"7fd513e5-0d7e-4684-89bc-9a3e21ce592c","resolution":{"observed_at":"2026-08-06T17:32:36.342972Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:35.886198Z","title":"Leveraging large language model for automatic patch correctness assessment","venue":null,"work_id":"7381dd86-441e-4f03-bf46-2e8f8be9dbd5","year":2024},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:34.165260Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:497ae9112f08b14344599091ed262b9c9264cf31c2e2dbdf615692a6f9233c7b","observation_id":"748fb1ef-1c7e-4a3a-a8da-615210c80fcb","resolution":{"observed_at":"2026-08-06T17:32:36.057972Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:35.582317Z","title":"Evaluating judges as evaluators: The JETTS benchmark of LLM-as-judges as test-time scaling evaluators","venue":null,"work_id":"d2f054ef-e77f-4a0b-9640-400356c07854","year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:34.238243Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:79969d7038c76f740a28751c4a2803c04d4435c74486d57bfa667092117b75a6","observation_id":"31548f35-18a2-4c74-b9fa-d3c5a84bfa18","resolution":{"observed_at":"2026-08-06T17:32:35.747660Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:35.314175Z","title":"JudgeLM: Fine-tuned large language models are scalable judges","venue":null,"work_id":"746bb14e-f04e-4bde-8ad2-9dfce059374f","year":2025},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:34.292050Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:292092fb6ef5dd24a34ffb323e1363e23fac955a8877e6b3749c875f4859062f","observation_id":"a1aa5af6-0d44-43b3-ba26-2d6a0bcf856e","resolution":{"observed_at":"2026-08-06T17:32:35.431422Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:32:41.383474Z","title":null,"venue":null,"work_id":"2862e88a-99db-465c-91e1-128a1691f13d","year":2023},"citing_paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks","version":2},"reference_index":1174,"source":"pdf_text","source_observed_at":"2026-08-06T17:32:31.633118Z"},"links":{"citing_paper":"/paper/2507.10535"},"observation_digest":"sha256:55a0e2f45aea69f641e55f6a348ae79ad6f17b3d226cc92b85cf4f6b80dfb8a7","observation_id":"c34bd653-6094-41c6-abab-36d6b78352cc","resolution":{"observed_at":"2026-08-06T17:32:41.495630Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.10535","last_updated":"2025-08-14T17:58:50Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-06T17:26:10.296563Z","submitted_at":"2025-07-14T17:56:29Z","title":"CodeJudgeBench: Benchmarking LLM-as-a-Judge for Coding Tasks"},"reference_resolution":{"displayed":61,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":29,"verified_exact":1,"verified_fuzzy":31},"total_outbound_references":61},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 61 of 61 outbound references and 11 inbound Pith citation observations for arXiv:2507.10535."}