{"as_of":"2026-08-06T19:28:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:15f9f1204bb5f90e24af439d81265d6019af8f44d43e3b211fd4712c37559e56","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":13,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":13,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":13,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":13,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T16:30:31.039397Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-01T14:05:46.736275Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2311.09835","last_updated":"2024-08-21T13:36:30Z","snapshot_observed_at":"2026-07-06T16:48:33.164024Z","submitted_at":"2023-11-16T12:03:21Z","title":"ML-Bench: Evaluating Large Language Models and Agents for Machine Learning Tasks on Repository-Level Code","version":5},"cited_work":{"arxiv_id":"2311.09835","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.09835","snapshot_observed_at":"2026-07-01T14:05:46.736275Z","title":"ML - Bench : Evaluating Large Language Models and Agents for Machine Learning Tasks on Repository - Level Code , August 2024","venue":null,"work_id":"c1ece405-4a98-4fe9-90a5-42fa6801339d","year":2023},"citing_paper":{"arxiv_id":"2308.00352","last_updated":"2024-11-01T14:36:52Z","snapshot_observed_at":"2026-07-06T16:01:07.532053Z","submitted_at":"2023-08-01T07:49:10Z","title":"MetaGPT: Meta Programming for A Multi-Agent Collaborative Framework","version":7},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-05-11T03:43:18.632292Z"},"links":{"cited_paper":"/paper/2311.09835","citing_paper":"/paper/2308.00352"},"observation_digest":"sha256:3bec1667b71724488f8e74e93732b9511275c49eab0a7048fcecab6b546635b3","observation_id":"ebf24730-2b79-4bb7-ac86-c752e9e58ab9","resolution":{"observed_at":"2026-05-11T03:43:18.933691Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.09835","last_updated":"2024-08-21T13:36:30Z","snapshot_observed_at":"2026-07-06T16:48:33.164024Z","submitted_at":"2023-11-16T12:03:21Z","title":"ML-Bench: Evaluating Large Language Models and Agents for Machine Learning Tasks on Repository-Level Code","version":5},"cited_work":{"arxiv_id":"2311.09835","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.09835","snapshot_observed_at":"2026-07-01T14:05:46.736275Z","title":"ML - Bench : Evaluating Large Language Models and Agents for Machine Learning Tasks on Repository - Level Code , August 2024","venue":null,"work_id":"c1ece405-4a98-4fe9-90a5-42fa6801339d","year":2023},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"cited_paper":"/paper/2311.09835","citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:6afbd633e5b9c5d2afbaf7a36940ed43c6ce633c36a5e43d5294d1b024bea5a8","observation_id":"230b2e79-42ba-4507-bf2f-bb3b8e302bd2","resolution":{"observed_at":"2026-05-23T19:13:21.535616Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.09835","last_updated":"2024-08-21T13:36:30Z","snapshot_observed_at":"2026-07-06T16:48:33.164024Z","submitted_at":"2023-11-16T12:03:21Z","title":"ML-Bench: Evaluating Large Language Models and Agents for Machine Learning Tasks on Repository-Level Code","version":5},"cited_work":{"arxiv_id":"2311.09835","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.09835","snapshot_observed_at":"2026-07-01T14:05:46.736275Z","title":"ML - Bench : Evaluating Large Language Models and Agents for Machine Learning Tasks on Repository - Level Code , August 2024","venue":null,"work_id":"c1ece405-4a98-4fe9-90a5-42fa6801339d","year":2023},"citing_paper":{"arxiv_id":"2506.04565","last_updated":"2026-05-08T15:50:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-05T02:34:43Z","title":"From Standalone LLMs to Integrated Intelligence: A Survey of Compound Al Systems","version":2},"reference_index":173,"source":"pdf_text","source_observed_at":"2026-05-19T11:49:36.574471Z"},"links":{"cited_paper":"/paper/2311.09835","citing_paper":"/paper/2506.04565"},"observation_digest":"sha256:0e521a7616bf3311e6464219d3830d16eada56a6231784a4ea7758113954eca5","observation_id":"0a3c29a7-66ad-42f9-8b46-abb6bc0485fa","resolution":{"observed_at":"2026-05-19T11:52:16.430624Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.09835","last_updated":"2024-08-21T13:36:30Z","snapshot_observed_at":"2026-07-06T16:48:33.164024Z","submitted_at":"2023-11-16T12:03:21Z","title":"ML-Bench: Evaluating Large Language Models and Agents for Machine Learning Tasks on Repository-Level Code","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.09835","snapshot_observed_at":"2026-08-06T16:30:31.039397Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.13300","last_updated":"2025-07-17T17:09:22Z","snapshot_observed_at":"2026-08-06T16:23:37.662421Z","submitted_at":"2025-07-17T17:09:22Z","title":"AbGen: Evaluating Large Language Models in Ablation Study Design and Evaluation for Scientific Research","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-06T16:30:31.039397Z"},"links":{"cited_paper":"/paper/2311.09835","citing_paper":"/paper/2507.13300"},"observation_digest":"sha256:86ebfca653f485cd6e1032be686fbb6e7aa635da3228907f5ffbd0060ca40dc8","observation_id":"c6f21ed6-1294-43ac-9006-709465b70578","resolution":{"observed_at":"2026-08-06T16:30:31.039397Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.09835","last_updated":"2024-08-21T13:36:30Z","snapshot_observed_at":"2026-07-06T16:48:33.164024Z","submitted_at":"2023-11-16T12:03:21Z","title":"ML-Bench: Evaluating Large Language Models and Agents for Machine Learning Tasks on Repository-Level Code","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.09835","snapshot_observed_at":"2026-08-05T12:24:02.277658Z","title":"Ml-bench: Evaluating large language models and agents for machine learning tasks on repository-level code","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.01684","last_updated":"2025-09-01T18:04:10Z","snapshot_observed_at":"2026-08-05T12:24:00.598679Z","submitted_at":"2025-09-01T18:04:10Z","title":"Reinforcement Learning for Machine Learning Engineering Agents","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-05T12:24:02.277658Z"},"links":{"cited_paper":"/paper/2311.09835","citing_paper":"/paper/2509.01684"},"observation_digest":"sha256:0c861e07600575c0e850854d1881bff8d7096aaf52ce46f6dfbc06846e561b02","observation_id":"744c909d-1d45-4ef6-aa4d-29e09ab1a074","resolution":{"observed_at":"2026-08-05T12:24:02.277658Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.09835","last_updated":"2024-08-21T13:36:30Z","snapshot_observed_at":"2026-07-06T16:48:33.164024Z","submitted_at":"2023-11-16T12:03:21Z","title":"ML-Bench: Evaluating Large Language Models and Agents for Machine Learning Tasks on Repository-Level Code","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.09835","snapshot_observed_at":"2026-08-02T23:31:16.475615Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.13723","last_updated":"2026-07-09T17:16:26Z","snapshot_observed_at":"2026-08-04T18:16:32.104416Z","submitted_at":"2026-02-14T11:07:58Z","title":"Compiling Large Multi-Modal Requirement Documents into Runnable Software Systems: From an Agentic Test-Driven Perspective","version":5},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-02T23:31:16.475615Z"},"links":{"cited_paper":"/paper/2311.09835","citing_paper":"/paper/2602.13723"},"observation_digest":"sha256:204a61a985a3a7610e16cad502352d04fc5dc2c4934278b45b76c342509a7466","observation_id":"174c2508-feb8-4a56-86a8-86a342144568","resolution":{"observed_at":"2026-08-02T23:31:16.475615Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.09835","last_updated":"2024-08-21T13:36:30Z","snapshot_observed_at":"2026-07-06T16:48:33.164024Z","submitted_at":"2023-11-16T12:03:21Z","title":"ML-Bench: Evaluating Large Language Models and Agents for Machine Learning Tasks on Repository-Level Code","version":5},"cited_work":{"arxiv_id":"2311.09835","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.09835","snapshot_observed_at":"2026-07-01T14:05:46.736275Z","title":"ML - Bench : Evaluating Large Language Models and Agents for Machine Learning Tasks on Repository - Level Code , August 2024","venue":null,"work_id":"c1ece405-4a98-4fe9-90a5-42fa6801339d","year":2023},"citing_paper":{"arxiv_id":"2605.08678","last_updated":"2026-07-06T13:36:13Z","snapshot_observed_at":"2026-07-31T14:41:52.251938Z","submitted_at":"2026-05-09T04:29:46Z","title":"MLS-Bench: A Holistic and Rigorous Assessment of AI Systems on Building Better AI","version":1},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-05-12T01:13:35.990078Z"},"links":{"cited_paper":"/paper/2311.09835","citing_paper":"/paper/2605.08678"},"observation_digest":"sha256:c14f37bdfb1c6e372ed07d4be519e4247b28da95847dcc78e5dc6c33af166f5f","observation_id":"18816a07-66c5-4ba8-a943-1a7a84a17885","resolution":{"observed_at":"2026-05-12T08:21:25.084754Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.09835","last_updated":"2024-08-21T13:36:30Z","snapshot_observed_at":"2026-07-06T16:48:33.164024Z","submitted_at":"2023-11-16T12:03:21Z","title":"ML-Bench: Evaluating Large Language Models and Agents for Machine Learning Tasks on Repository-Level Code","version":5},"cited_work":{"arxiv_id":"2311.09835","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.09835","snapshot_observed_at":"2026-07-01T14:05:46.736275Z","title":"ML - Bench : Evaluating Large Language Models and Agents for Machine Learning Tasks on Repository - Level Code , August 2024","venue":null,"work_id":"c1ece405-4a98-4fe9-90a5-42fa6801339d","year":2023},"citing_paper":{"arxiv_id":"2605.08678","last_updated":"2026-07-06T13:36:13Z","snapshot_observed_at":"2026-07-31T14:41:52.251938Z","submitted_at":"2026-05-09T04:29:46Z","title":"MLS-Bench: A Holistic and Rigorous Assessment of AI Systems on Building Better AI","version":2},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-06-30T23:12:57.154537Z"},"links":{"cited_paper":"/paper/2311.09835","citing_paper":"/paper/2605.08678"},"observation_digest":"sha256:03312f65066bc71782df3b5835181e97b26761a2d3d78398b21a55f27239767f","observation_id":"cecd6380-a041-4259-aa06-cedd57d3e95b","resolution":{"observed_at":"2026-07-01T13:25:46.005528Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.09835","last_updated":"2024-08-21T13:36:30Z","snapshot_observed_at":"2026-07-06T16:48:33.164024Z","submitted_at":"2023-11-16T12:03:21Z","title":"ML-Bench: Evaluating Large Language Models and Agents for Machine Learning Tasks on Repository-Level Code","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.09835","snapshot_observed_at":"2026-07-12T17:14:49.310598Z","title":"Ml-bench: Evaluating large language models and agents for machine learning tasks on repository-level code.arXiv preprint arXiv:2311.09835, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2605.08678","last_updated":"2026-07-06T13:36:13Z","snapshot_observed_at":"2026-07-31T14:41:52.251938Z","submitted_at":"2026-05-09T04:29:46Z","title":"MLS-Bench: A Holistic and Rigorous Assessment of AI Systems on Building Better AI","version":3},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-07-12T17:14:49.310598Z"},"links":{"cited_paper":"/paper/2311.09835","citing_paper":"/paper/2605.08678"},"observation_digest":"sha256:af125c45cfe84617bd16b1ba2da93b30e9a843ab33156f969b079581724d2dd1","observation_id":"ef75ebca-0ea7-44ac-8c98-c58ed522f8cb","resolution":{"observed_at":"2026-07-12T17:14:49.310598Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.09835","last_updated":"2024-08-21T13:36:30Z","snapshot_observed_at":"2026-07-06T16:48:33.164024Z","submitted_at":"2023-11-16T12:03:21Z","title":"ML-Bench: Evaluating Large Language Models and Agents for Machine Learning Tasks on Repository-Level Code","version":5},"cited_work":{"arxiv_id":"2311.09835","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.09835","snapshot_observed_at":"2026-07-01T14:05:46.736275Z","title":"ML - Bench : Evaluating Large Language Models and Agents for Machine Learning Tasks on Repository - Level Code , August 2024","venue":null,"work_id":"c1ece405-4a98-4fe9-90a5-42fa6801339d","year":2023},"citing_paper":{"arxiv_id":"2605.12376","last_updated":"2026-06-04T17:58:18Z","snapshot_observed_at":"2026-07-06T23:24:03.984179Z","submitted_at":"2026-05-12T16:42:38Z","title":"ProfiliTable: Profiling-Driven Tabular Data Processing via Agentic Workflows","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-13T05:56:36.312877Z"},"links":{"cited_paper":"/paper/2311.09835","citing_paper":"/paper/2605.12376"},"observation_digest":"sha256:7006528107a5ebd2c4115cad389da216aea3b96ad842107c55acea6dcbeb02f4","observation_id":"637a3010-188f-4e81-8dbc-ff3578f8ba1d","resolution":{"observed_at":"2026-05-13T05:57:22.501265Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.09835","last_updated":"2024-08-21T13:36:30Z","snapshot_observed_at":"2026-07-06T16:48:33.164024Z","submitted_at":"2023-11-16T12:03:21Z","title":"ML-Bench: Evaluating Large Language Models and Agents for Machine Learning Tasks on Repository-Level Code","version":5},"cited_work":{"arxiv_id":"2311.09835","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.09835","snapshot_observed_at":"2026-07-01T14:05:46.736275Z","title":"ML - Bench : Evaluating Large Language Models and Agents for Machine Learning Tasks on Repository - Level Code , August 2024","venue":null,"work_id":"c1ece405-4a98-4fe9-90a5-42fa6801339d","year":2023},"citing_paper":{"arxiv_id":"2605.12376","last_updated":"2026-06-04T17:58:18Z","snapshot_observed_at":"2026-07-06T23:24:03.984179Z","submitted_at":"2026-05-12T16:42:38Z","title":"ProfiliTable: Profiling-Driven Tabular Data Processing via Agentic Workflows","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-06-30T22:18:45.189576Z"},"links":{"cited_paper":"/paper/2311.09835","citing_paper":"/paper/2605.12376"},"observation_digest":"sha256:56f3e3b8769f908bf597ff67337444e89aea378f1735ddaf94f537e59cfefc5d","observation_id":"26cc257b-1017-4935-887b-1bbe3b2806a1","resolution":{"observed_at":"2026-07-01T14:05:46.737909Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.09835","last_updated":"2024-08-21T13:36:30Z","snapshot_observed_at":"2026-07-06T16:48:33.164024Z","submitted_at":"2023-11-16T12:03:21Z","title":"ML-Bench: Evaluating Large Language Models and Agents for Machine Learning Tasks on Repository-Level Code","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.09835","snapshot_observed_at":"2026-07-31T01:39:48.754483Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.25090","last_updated":"2026-07-27T21:30:39Z","snapshot_observed_at":"2026-08-03T01:32:45.260967Z","submitted_at":"2026-07-27T21:30:39Z","title":"Matryoshka Agent: Unfolding Sub-Agents for Long-Horizon Machine Learning Engineering","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-07-31T01:39:48.754483Z"},"links":{"cited_paper":"/paper/2311.09835","citing_paper":"/paper/2607.25090"},"observation_digest":"sha256:f9ba6f04d3bb810695207d3c9903d580f4a158d872323887df778e074fbf0eef","observation_id":"3e1dffac-109e-4453-9e1c-48b967604fc4","resolution":{"observed_at":"2026-07-31T01:39:48.754483Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.09835","last_updated":"2024-08-21T13:36:30Z","snapshot_observed_at":"2026-07-06T16:48:33.164024Z","submitted_at":"2023-11-16T12:03:21Z","title":"ML-Bench: Evaluating Large Language Models and Agents for Machine Learning Tasks on Repository-Level Code","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.09835","snapshot_observed_at":"2026-08-03T03:15:14.160911Z","title":"ML-Bench: Evaluating large language models and agents for machine learning tasks on repository-level code, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29626","last_updated":"2026-07-31T16:58:00Z","snapshot_observed_at":"2026-08-05T23:16:09.400110Z","submitted_at":"2026-07-31T16:58:00Z","title":"AgentHPOBench: A Benchmark For Evaluating LLM Agents as Sequential Hyperparameter Optimizers","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-03T03:15:14.160911Z"},"links":{"cited_paper":"/paper/2311.09835","citing_paper":"/paper/2607.29626"},"observation_digest":"sha256:e3abe04f4c3ad4fe67cd0f95522ecb931f6a9d485410c4fd715ddad3f6bb6984","observation_id":"b8946c5c-24df-47d6-8697-164497746df3","resolution":{"observed_at":"2026-08-03T03:15:14.160911Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2311.09835/citation-record","integrity":"/paper/2311.09835/integrity","json":"/paper/2311.09835/citation-record.json","paper":"/paper/2311.09835"},"outbound":[],"paper":{"arxiv_id":"2311.09835","last_updated":"2024-08-21T13:36:30Z","latest_version":5,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T16:48:33.164024Z","submitted_at":"2023-11-16T12:03:21Z","title":"ML-Bench: Evaluating Large Language Models and Agents for Machine Learning Tasks on Repository-Level Code"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 13 inbound Pith citation observations for arXiv:2311.09835."}