{"as_of":"2026-08-10T17:36:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:0c46dd44e77b89bef8f45786d7a75d1af5888396a81e9b70a2019443e5340059","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":11,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":11,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":11,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":11,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T04:33:10.690842Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2407.11470","last_updated":"2024-10-09T05:59:07Z","snapshot_observed_at":"2026-07-06T18:47:00.524107Z","submitted_at":"2024-07-16T08:08:48Z","title":"Beyond Correctness: Benchmarking Multi-dimensional Code Generation for Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.11470","snapshot_observed_at":"2026-08-07T04:33:10.690842Z","title":"Beyond correctness: Benchmarking multi-dimensional code generation for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10481","last_updated":"2025-06-12T08:33:38Z","snapshot_observed_at":"2026-08-08T11:51:34.890409Z","submitted_at":"2025-06-12T08:33:38Z","title":"OIBench: Benchmarking Strong Reasoning Models with Olympiad in Informatics","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T04:33:10.690842Z"},"links":{"cited_paper":"/paper/2407.11470","citing_paper":"/paper/2506.10481"},"observation_digest":"sha256:b29c66890ef310738b532354ad980152fc826dfc15423497cd503eff8d07af67","observation_id":"936321a4-c108-462a-bd8c-590b38ec447c","resolution":{"observed_at":"2026-08-07T04:33:10.690842Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.11470","last_updated":"2024-10-09T05:59:07Z","snapshot_observed_at":"2026-07-06T18:47:00.524107Z","submitted_at":"2024-07-16T08:08:48Z","title":"Beyond Correctness: Benchmarking Multi-dimensional Code Generation for Large Language Models","version":2},"cited_work":{"arxiv_id":"2407.11470","doi":"10.48550/arxiv.2407.11470","metadata_source":"pith","pith_arxiv_id":"2407.11470","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Beyond correctness: Benchmarking multi-dimensional code generation for large language models","venue":"cs.SE","work_id":"3e3db54f-62ff-456d-8fbb-e697fe3027e5","year":2024},"citing_paper":{"arxiv_id":"2507.11687","last_updated":"2026-04-20T02:18:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-15T19:44:20Z","title":"MetaLint: Easy-to-Hard Generalization for Code Linting","version":4},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-05-19T04:07:31.283348Z"},"links":{"cited_paper":"/paper/2407.11470","citing_paper":"/paper/2507.11687"},"observation_digest":"sha256:73290e9fb66de04101fb55e4197a42f593321c201480699b9723506e61ff3cea","observation_id":"d2902e08-b1bb-4dfd-8b28-1bae3b13dc2d","resolution":{"observed_at":"2026-05-19T04:12:02.872408Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.11470","last_updated":"2024-10-09T05:59:07Z","snapshot_observed_at":"2026-07-06T18:47:00.524107Z","submitted_at":"2024-07-16T08:08:48Z","title":"Beyond Correctness: Benchmarking Multi-dimensional Code Generation for Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.11470","snapshot_observed_at":"2026-08-05T18:40:05.431137Z","title":"Beyond correctness: Benchmarking multi-dimensional code generation for large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.14419","last_updated":"2025-08-20T04:31:31Z","snapshot_observed_at":"2026-08-06T07:38:55.270700Z","submitted_at":"2025-08-20T04:31:31Z","title":"Static Analysis as a Feedback Loop: Enhancing LLM-Generated Code Beyond Correctness","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-05T18:40:05.431137Z"},"links":{"cited_paper":"/paper/2407.11470","citing_paper":"/paper/2508.14419"},"observation_digest":"sha256:5a4526898b343a8d9fd990862f6651991470641f78fde437924600a1d3d7a05f","observation_id":"e031d502-458c-4d91-bdb1-dc320af3c388","resolution":{"observed_at":"2026-08-05T18:40:05.431137Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.11470","last_updated":"2024-10-09T05:59:07Z","snapshot_observed_at":"2026-07-06T18:47:00.524107Z","submitted_at":"2024-07-16T08:08:48Z","title":"Beyond Correctness: Benchmarking Multi-dimensional Code Generation for Large Language Models","version":2},"cited_work":{"arxiv_id":"2407.11470","doi":"10.48550/arxiv.2407.11470","metadata_source":"pith","pith_arxiv_id":"2407.11470","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Beyond correctness: Benchmarking multi-dimensional code generation for large language models","venue":"cs.SE","work_id":"3e3db54f-62ff-456d-8fbb-e697fe3027e5","year":2024},"citing_paper":{"arxiv_id":"2510.15494","last_updated":"2026-04-09T09:01:12Z","snapshot_observed_at":"2026-08-06T20:17:52.315302Z","submitted_at":"2025-10-17T10:06:52Z","title":"Do AI Models Dream of Faster Code? An Empirical Study on LLM-Proposed Performance Improvements in Real-World Software","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-18T06:39:42.391102Z"},"links":{"cited_paper":"/paper/2407.11470","citing_paper":"/paper/2510.15494"},"observation_digest":"sha256:1e57e0afe60bf3ee2776f3d28dc2cbb2169b5539acc2bd55cee2336347821ca5","observation_id":"d1d97fc7-ad59-4126-a863-8e9ed0db2a6e","resolution":{"observed_at":"2026-05-18T06:41:00.347065Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.11470","last_updated":"2024-10-09T05:59:07Z","snapshot_observed_at":"2026-07-06T18:47:00.524107Z","submitted_at":"2024-07-16T08:08:48Z","title":"Beyond Correctness: Benchmarking Multi-dimensional Code Generation for Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.11470","snapshot_observed_at":"2026-08-04T06:49:11.634098Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2511.15817","last_updated":"2026-08-03T14:58:52Z","snapshot_observed_at":"2026-08-08T06:29:32.806927Z","submitted_at":"2025-11-19T19:18:28Z","title":"A Causal Perspective on Measuring, Explaining and Mitigating Smells in LLM-Generated Code","version":6},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-04T06:49:11.634098Z"},"links":{"cited_paper":"/paper/2407.11470","citing_paper":"/paper/2511.15817"},"observation_digest":"sha256:75b4bb6c87f70e61b4d1c9abd50c18789ae547fc544eae306b40ef0115c5d0dd","observation_id":"dac42bef-5b25-4d36-b2f1-8906c18b8a89","resolution":{"observed_at":"2026-08-04T06:49:11.634098Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.11470","last_updated":"2024-10-09T05:59:07Z","snapshot_observed_at":"2026-07-06T18:47:00.524107Z","submitted_at":"2024-07-16T08:08:48Z","title":"Beyond Correctness: Benchmarking Multi-dimensional Code Generation for Large Language Models","version":2},"cited_work":{"arxiv_id":"2407.11470","doi":"10.48550/arxiv.2407.11470","metadata_source":"pith","pith_arxiv_id":"2407.11470","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Beyond correctness: Benchmarking multi-dimensional code generation for large language models","venue":"cs.SE","work_id":"3e3db54f-62ff-456d-8fbb-e697fe3027e5","year":2024},"citing_paper":{"arxiv_id":"2605.05267","last_updated":"2026-05-06T09:38:31Z","snapshot_observed_at":"2026-08-04T17:41:12.164736Z","submitted_at":"2026-05-06T09:38:31Z","title":"Bridging Generation and Training: A Systematic Review of Quality Issues in LLMs for Code","version":1},"reference_index":154,"source":"pdf_text","source_observed_at":"2026-05-08T17:37:51.790000Z"},"links":{"cited_paper":"/paper/2407.11470","citing_paper":"/paper/2605.05267"},"observation_digest":"sha256:f927bd6a11bf29a0234875aa6cea790e6d75cc1b36a906e7aadc0258f5e52e46","observation_id":"b722956f-ca0a-477b-8cc0-a3193382c056","resolution":{"observed_at":"2026-05-11T17:21:10.766568Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.11470","last_updated":"2024-10-09T05:59:07Z","snapshot_observed_at":"2026-07-06T18:47:00.524107Z","submitted_at":"2024-07-16T08:08:48Z","title":"Beyond Correctness: Benchmarking Multi-dimensional Code Generation for Large Language Models","version":2},"cited_work":{"arxiv_id":"2407.11470","doi":"10.48550/arxiv.2407.11470","metadata_source":"pith","pith_arxiv_id":"2407.11470","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Beyond correctness: Benchmarking multi-dimensional code generation for large language models","venue":"cs.SE","work_id":"3e3db54f-62ff-456d-8fbb-e697fe3027e5","year":2024},"citing_paper":{"arxiv_id":"2605.13776","last_updated":"2026-05-13T16:54:51Z","snapshot_observed_at":"2026-07-06T23:25:16.229144Z","submitted_at":"2026-05-13T16:54:51Z","title":"\"Like Taking the Path of Least Resistance\": Exploring the Impact of LLM Interaction on the Creative Process of Programming","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-05-14T17:40:19.994225Z"},"links":{"cited_paper":"/paper/2407.11470","citing_paper":"/paper/2605.13776"},"observation_digest":"sha256:2c4055e4a54bacb70cf1c5de5b7802bb44938917eec69e45f6b9c99e8c2d101a","observation_id":"8743e843-2620-47c1-afb5-b5b82e95057e","resolution":{"observed_at":"2026-05-14T17:42:30.623598Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.11470","last_updated":"2024-10-09T05:59:07Z","snapshot_observed_at":"2026-07-06T18:47:00.524107Z","submitted_at":"2024-07-16T08:08:48Z","title":"Beyond Correctness: Benchmarking Multi-dimensional Code Generation for Large Language Models","version":2},"cited_work":{"arxiv_id":"2407.11470","doi":"10.48550/arxiv.2407.11470","metadata_source":"pith","pith_arxiv_id":"2407.11470","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Beyond correctness: Benchmarking multi-dimensional code generation for large language models","venue":"cs.SE","work_id":"3e3db54f-62ff-456d-8fbb-e697fe3027e5","year":2024},"citing_paper":{"arxiv_id":"2605.25296","last_updated":"2026-05-24T23:20:12Z","snapshot_observed_at":"2026-07-06T23:35:16.647496Z","submitted_at":"2026-05-24T23:20:12Z","title":"Subjective Code Preferences in Experts and Large Language Models","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-29T23:18:27.778928Z"},"links":{"cited_paper":"/paper/2407.11470","citing_paper":"/paper/2605.25296"},"observation_digest":"sha256:25b394313eaf0dd7853f53e3dd2185a14525be2ac2694c77137d02f37070ee97","observation_id":"d49ddea4-c397-4d87-bc76-a4c43edbef35","resolution":{"observed_at":"2026-06-29T23:24:01.812357Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.11470","last_updated":"2024-10-09T05:59:07Z","snapshot_observed_at":"2026-07-06T18:47:00.524107Z","submitted_at":"2024-07-16T08:08:48Z","title":"Beyond Correctness: Benchmarking Multi-dimensional Code Generation for Large Language Models","version":2},"cited_work":{"arxiv_id":"2407.11470","doi":"10.48550/arxiv.2407.11470","metadata_source":"pith","pith_arxiv_id":"2407.11470","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Beyond correctness: Benchmarking multi-dimensional code generation for large language models","venue":"cs.SE","work_id":"3e3db54f-62ff-456d-8fbb-e697fe3027e5","year":2024},"citing_paper":{"arxiv_id":"2606.08588","last_updated":"2026-06-07T12:01:44Z","snapshot_observed_at":"2026-08-03T20:06:31.609958Z","submitted_at":"2026-06-07T12:01:44Z","title":"LLM vs. Human Unit Tests: Fault Detection on Real Python Bugs","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-27T18:07:02.131365Z"},"links":{"cited_paper":"/paper/2407.11470","citing_paper":"/paper/2606.08588"},"observation_digest":"sha256:8a54aa167c8d76e33a11b611703859f6b2bc7e1b7201ee9baaea427aac5b82f0","observation_id":"8a716987-56df-4609-b807-15fdae804c12","resolution":{"observed_at":"2026-07-02T23:37:27.555356Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.11470","last_updated":"2024-10-09T05:59:07Z","snapshot_observed_at":"2026-07-06T18:47:00.524107Z","submitted_at":"2024-07-16T08:08:48Z","title":"Beyond Correctness: Benchmarking Multi-dimensional Code Generation for Large Language Models","version":2},"cited_work":{"arxiv_id":"2407.11470","doi":"10.48550/arxiv.2407.11470","metadata_source":"pith","pith_arxiv_id":"2407.11470","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Beyond correctness: Benchmarking multi-dimensional code generation for large language models","venue":"cs.SE","work_id":"3e3db54f-62ff-456d-8fbb-e697fe3027e5","year":2024},"citing_paper":{"arxiv_id":"2607.08009","last_updated":"2026-07-09T00:27:07Z","snapshot_observed_at":"2026-08-03T01:18:12.663488Z","submitted_at":"2026-07-09T00:27:07Z","title":"From Execution to Education: A Bloom-Aligned Framework for Measuring Educational Control in LLMs","version":1},"reference_index":206,"source":"arxiv_source","source_observed_at":"2026-07-10T13:49:17.343893Z"},"links":{"cited_paper":"/paper/2407.11470","citing_paper":"/paper/2607.08009"},"observation_digest":"sha256:0c6f1d9ab1e29cfeaba9f17fac7060f467f482627a17e6327311ba4f381d7b78","observation_id":"4f3dcc34-e852-46e0-b251-62dd46f8d9c2","resolution":{"observed_at":"2026-07-10T13:57:06.898375Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.11470","last_updated":"2024-10-09T05:59:07Z","snapshot_observed_at":"2026-07-06T18:47:00.524107Z","submitted_at":"2024-07-16T08:08:48Z","title":"Beyond Correctness: Benchmarking Multi-dimensional Code Generation for Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.11470","snapshot_observed_at":"2026-08-05T17:14:07.898036Z","title":"arXiv preprint arXiv:2407.11470 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.03535","last_updated":"2026-08-04T12:15:17Z","snapshot_observed_at":"2026-08-10T03:38:58.434071Z","submitted_at":"2026-08-04T12:15:17Z","title":"CodeAssay: A Multi-Metric Benchmark with Audited Ground Truth for LLM Code Generation","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-05T17:14:07.898036Z"},"links":{"cited_paper":"/paper/2407.11470","citing_paper":"/paper/2608.03535"},"observation_digest":"sha256:d1d5609df3062c8b15cd5846d0998ab79bee7fd10439ab94ff1385a5fa42b75b","observation_id":"de004fe9-a825-4661-90da-7af60fb827bf","resolution":{"observed_at":"2026-08-05T17:14:07.898036Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2407.11470/citation-record","integrity":"/paper/2407.11470/integrity","json":"/paper/2407.11470/citation-record.json","paper":"/paper/2407.11470"},"outbound":[],"paper":{"arxiv_id":"2407.11470","last_updated":"2024-10-09T05:59:07Z","latest_version":2,"primary_category":"cs.SE","snapshot_observed_at":"2026-07-06T18:47:00.524107Z","submitted_at":"2024-07-16T08:08:48Z","title":"Beyond Correctness: Benchmarking Multi-dimensional Code Generation for Large Language Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 11 inbound Pith citation observations for arXiv:2407.11470."}