{"as_of":"2026-08-07T10:32:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:d5f0ea95db86664ea5a5b3ba37d9bcd5fe6c19350cb95f0fbd966db430353832","coverage":[{"denominator":45,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":45,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T04:23:50.683069Z","state":"measured"},{"denominator":53,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":53,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":8,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":8,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T14:04:57.314444Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T20:20:07.360446Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"cited_work":{"arxiv_id":"2506.10764","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.10764","snapshot_observed_at":"2026-07-04T20:20:07.360446Z","title":"arXiv preprint arXiv:2506.10764 (2025)","venue":null,"work_id":"ab07a44a-de2d-40af-aa9d-042b7ba635ed","year":2025},"citing_paper":{"arxiv_id":"2604.19440","last_updated":"2026-04-21T13:16:45Z","snapshot_observed_at":"2026-08-03T04:28:53.538807Z","submitted_at":"2026-04-21T13:16:45Z","title":"What Makes an LLM a Good Optimizer? A Trajectory Analysis of LLM-Guided Evolutionary Search","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T02:19:18.121220Z"},"links":{"cited_paper":"/paper/2506.10764","citing_paper":"/paper/2604.19440"},"observation_digest":"sha256:0fe45ad92a1b76fef313394c262a8c2e6a27f47af43a628cd47723f5b6932989","observation_id":"d1d85105-0d9e-478a-b365-51b8874b8d35","resolution":{"observed_at":"2026-05-11T13:06:05.611346Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"cited_work":{"arxiv_id":"2506.10764","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.10764","snapshot_observed_at":"2026-07-04T20:20:07.360446Z","title":"arXiv preprint arXiv:2506.10764 (2025)","venue":null,"work_id":"ab07a44a-de2d-40af-aa9d-042b7ba635ed","year":2025},"citing_paper":{"arxiv_id":"2605.08905","last_updated":"2026-05-09T11:57:25Z","snapshot_observed_at":"2026-07-06T23:21:06.773761Z","submitted_at":"2026-05-09T11:57:25Z","title":"Forge: Quality-Aware Reinforcement Learning for NP-Hard Optimization in LLMs","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-05-12T02:44:33.143247Z"},"links":{"cited_paper":"/paper/2506.10764","citing_paper":"/paper/2605.08905"},"observation_digest":"sha256:1c8cfce763e9af93bb17ac78d5233b55a9439031dcb11b3fca3b432cfce531ed","observation_id":"adec540f-a8ac-440d-a6f2-7f273a7efedc","resolution":{"observed_at":"2026-05-12T02:46:18.873851Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"cited_work":{"arxiv_id":"2506.10764","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.10764","snapshot_observed_at":"2026-07-04T20:20:07.360446Z","title":"arXiv preprint arXiv:2506.10764 (2025)","venue":null,"work_id":"ab07a44a-de2d-40af-aa9d-042b7ba635ed","year":2025},"citing_paper":{"arxiv_id":"2605.19447","last_updated":"2026-05-19T07:00:55Z","snapshot_observed_at":"2026-08-02T09:08:33.207894Z","submitted_at":"2026-05-19T07:00:55Z","title":"What and When to Distill: Selective Hindsight Distillation for Multi-Turn Agents","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-20T05:41:23.712146Z"},"links":{"cited_paper":"/paper/2506.10764","citing_paper":"/paper/2605.19447"},"observation_digest":"sha256:e1926183d2e0c702e5f38090609b284f262f739abff1dd7b9c9dc859a97de184","observation_id":"d9b968af-82cf-4ae5-90d6-7438f9d985eb","resolution":{"observed_at":"2026-05-20T05:43:05.828295Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"cited_work":{"arxiv_id":"2506.10764","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.10764","snapshot_observed_at":"2026-07-04T20:20:07.360446Z","title":"arXiv preprint arXiv:2506.10764 (2025)","venue":null,"work_id":"ab07a44a-de2d-40af-aa9d-042b7ba635ed","year":2025},"citing_paper":{"arxiv_id":"2605.20849","last_updated":"2026-05-20T07:40:05Z","snapshot_observed_at":"2026-08-05T14:32:21.145968Z","submitted_at":"2026-05-20T07:40:05Z","title":"Large Language Models for Operations Research: A Comprehensive Survey","version":1},"reference_index":174,"source":"pdf_text","source_observed_at":"2026-05-21T03:56:29.983335Z"},"links":{"cited_paper":"/paper/2506.10764","citing_paper":"/paper/2605.20849"},"observation_digest":"sha256:7b3dc0898d5b7b21e7b888b714416238510f85485ce4ddfbb424fa1284ec6180","observation_id":"3b67f314-ccd9-4705-bca2-1bc82a5318f9","resolution":{"observed_at":"2026-05-21T03:59:32.604852Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"cited_work":{"arxiv_id":"2506.10764","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.10764","snapshot_observed_at":"2026-07-04T20:20:07.360446Z","title":"arXiv preprint arXiv:2506.10764 (2025)","venue":null,"work_id":"ab07a44a-de2d-40af-aa9d-042b7ba635ed","year":2025},"citing_paper":{"arxiv_id":"2606.19338","last_updated":"2026-06-17T17:59:34Z","snapshot_observed_at":"2026-07-06T23:54:37.596257Z","submitted_at":"2026-06-17T17:59:34Z","title":"Beyond the Current Observation: Evaluating Multimodal Large Language Models in Controllable Non-Markov Games","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-06-26T21:17:02.332687Z"},"links":{"cited_paper":"/paper/2506.10764","citing_paper":"/paper/2606.19338"},"observation_digest":"sha256:64ea1b1980ea89f32bd74f3c3911ff37c4962c8ef7171bcd559c79a0334ee9ad","observation_id":"1e605860-bb3a-4446-bdb3-38aadd4f87eb","resolution":{"observed_at":"2026-07-04T00:19:13.672489Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"cited_work":{"arxiv_id":"2506.10764","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.10764","snapshot_observed_at":"2026-07-04T20:20:07.360446Z","title":"arXiv preprint arXiv:2506.10764 (2025)","venue":null,"work_id":"ab07a44a-de2d-40af-aa9d-042b7ba635ed","year":2025},"citing_paper":{"arxiv_id":"2606.25832","last_updated":"2026-06-25T05:38:04Z","snapshot_observed_at":"2026-08-03T01:16:19.684931Z","submitted_at":"2026-06-24T13:48:06Z","title":"MiniOpt: Reasoning to Model and Solve General Optimization Problems with Limited Resources","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-25T20:19:30.720291Z"},"links":{"cited_paper":"/paper/2506.10764","citing_paper":"/paper/2606.25832"},"observation_digest":"sha256:dd889c57418d7a4dc99687781eb11aa3cc5eddf594307c31cff0c602406ea6c1","observation_id":"a99942cd-2771-402e-bc91-67b96f67ca3e","resolution":{"observed_at":"2026-07-04T20:20:07.361970Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"cited_work":{"arxiv_id":"2506.10764","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.10764","snapshot_observed_at":"2026-07-04T20:20:07.360446Z","title":"arXiv preprint arXiv:2506.10764 (2025)","venue":null,"work_id":"ab07a44a-de2d-40af-aa9d-042b7ba635ed","year":2025},"citing_paper":{"arxiv_id":"2606.25832","last_updated":"2026-06-25T05:38:04Z","snapshot_observed_at":"2026-08-03T01:16:19.684931Z","submitted_at":"2026-06-24T13:48:06Z","title":"MiniOpt: Reasoning to Model and Solve General Optimization Problems with Limited Resources","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-26T05:18:55.074710Z"},"links":{"cited_paper":"/paper/2506.10764","citing_paper":"/paper/2606.25832"},"observation_digest":"sha256:c07cff030c4052faec50a116283debe15cbbcdad7b6b14e17dd3635de67d0a80","observation_id":"0d3bf3f0-e0b5-4632-93cb-a063009b55b1","resolution":{"observed_at":"2026-07-04T13:19:50.751708Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.10764","snapshot_observed_at":"2026-08-02T14:04:57.314444Z","title":"doi:10.48550/arXiv.2506.10764 , abstract =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.18256","last_updated":"2026-05-14T06:47:20Z","snapshot_observed_at":"2026-08-05T02:13:07.767245Z","submitted_at":"2026-05-14T06:47:20Z","title":"PEARL: Solver-in-the-Loop Interactive Optimization Modeling from Natural Language","version":1},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-08-02T14:04:57.314444Z"},"links":{"cited_paper":"/paper/2506.10764","citing_paper":"/paper/2607.18256"},"observation_digest":"sha256:50dc78f191863f6a4f61c84076e0e5a5f59a6e894e6a0e109324e1659e1f3145","observation_id":"a9ffee4b-2b47-4112-8e2e-8d49e0cf63f5","resolution":{"observed_at":"2026-08-02T14:04:57.314444Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2506.10764/citation-record","integrity":"/paper/2506.10764/integrity","json":"/paper/2506.10764/citation-record.json","paper":"/paper/2506.10764"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-07T04:23:50.543505Z","title":"Gpt-4 technical report.arXiv preprint arXiv:2303.08774, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.543505Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:52fe921d4e17ac3576c29ffce30e4b0be23052d14fa7325160eeea9801810c70","observation_id":"5b6495d8-7106-4e4e-9e68-c534f22f34d1","resolution":{"observed_at":"2026-08-07T04:23:50.543505Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1905.13319","last_updated":"2019-05-30T21:28:12Z","snapshot_observed_at":"2026-07-06T07:56:54.143277Z","submitted_at":"2019-05-30T21:28:12Z","title":"MathQA: Towards Interpretable Math Word Problem Solving with Operation-Based Formalisms","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1905.13319","snapshot_observed_at":"2026-08-07T04:23:50.547360Z","title":"Mathqa: Towards interpretable math word problem solving with operation-based formalisms.arXiv preprint arXiv:1905.13319, 2019","venue":null,"work_id":null,"year":1905},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.547360Z"},"links":{"cited_paper":"/paper/1905.13319","citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:3bc9d08000893c93220bb5e9bd7e52f975573fbe9b6e503d5f2e0575229db23b","observation_id":"692304e6-190e-4b51-affa-5494d3e16856","resolution":{"observed_at":"2026-08-07T04:23:50.547360Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2108.07732","last_updated":"2021-08-16T03:57:30Z","snapshot_observed_at":"2026-08-02T19:23:53.535075Z","submitted_at":"2021-08-16T03:57:30Z","title":"Program Synthesis with Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2108.07732","snapshot_observed_at":"2026-08-07T04:23:50.551498Z","title":"Program synthesis with large language models.arXiv preprint arXiv:2108.07732, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.551498Z"},"links":{"cited_paper":"/paper/2108.07732","citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:d13904b60ed49e11a4ddaa6b76e8b8bde81b5e58ebc97457a7b9687761c9fff5","observation_id":"2e969ac0-cb4b-40eb-82a3-f510cb7984d6","resolution":{"observed_at":"2026-08-07T04:23:50.551498Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:23:50.554887Z","title":"Language models are few-shot learners.Advances in neural information processing systems, 33:1877–1901, 2020","venue":null,"work_id":null,"year":1901},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.554887Z"},"links":{"citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:54aa407ccb3a2649cb30a21a6cfe16fce7bba6981c231baf7b2386278cd9bcd9","observation_id":"39138021-4c0d-44eb-af22-74039c8e02bb","resolution":{"observed_at":"2026-08-07T04:23:50.554887Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2107.03374","last_updated":"2021-07-14T17:16:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-07-07T17:41:24Z","title":"Evaluating Large Language Models Trained on Code","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2107.03374","snapshot_observed_at":"2026-08-07T04:23:50.561890Z","title":"Evaluating large language models trained on code.arXiv preprint arXiv:2107.03374, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.561890Z"},"links":{"cited_paper":"/paper/2107.03374","citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:e3233b51dc51a7eafcf3d60f599c5d5c7845533c9fd6a0459c468b8fb54ca13e","observation_id":"895e5a4d-eca7-4c88-ba1b-70ad41bb08db","resolution":{"observed_at":"2026-08-07T04:23:50.561890Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:23:50.565288Z","title":"Palm: Scaling language modeling with pathways.Journal of Machine Learning Research, 24(240):1– 113, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.565288Z"},"links":{"citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:dbd87da8fc6770db88fa3d980204fd5d9df6c576f40bfd762b2f55f3719e8ead","observation_id":"d36b7c62-4ba5-4106-b181-07d11b280735","resolution":{"observed_at":"2026-08-07T04:23:50.565288Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-07T01:45:38.840969Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-07T04:23:50.568987Z","title":"Training verifiers to solve math word problems.arXiv preprint arXiv:2110.14168, 2021","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.568987Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:65e99df08bc145d033c39c121183124a3435c64d5eedd5d9c3802becd5ec886d","observation_id":"4bbc9d8e-a657-4e40-8e9a-35c1b8918355","resolution":{"observed_at":"2026-08-07T04:23:50.568987Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.09712","last_updated":"2022-05-19T17:25:28Z","snapshot_observed_at":"2026-08-05T16:13:51.965874Z","submitted_at":"2022-05-19T17:25:28Z","title":"Selection-Inference: Exploiting Large Language Models for Interpretable Logical Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.09712","snapshot_observed_at":"2026-08-07T04:23:50.571999Z","title":"Selection-inference: Exploiting large language models for interpretable logical reasoning.arXiv preprint arXiv:2205.09712, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.571999Z"},"links":{"cited_paper":"/paper/2205.09712","citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:cf809d64309f730b5036393225dba615b99f6d0760d2576b4ee05615edb0a331","observation_id":"ac041da5-8e76-4147-b5dc-be0ffe382590","resolution":{"observed_at":"2026-08-07T04:23:50.571999Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.14890","last_updated":"2024-02-12T17:30:25Z","snapshot_observed_at":"2026-07-06T17:07:19.721701Z","submitted_at":"2023-12-22T18:07:44Z","title":"NPHardEval: Dynamic Benchmark on Reasoning Ability of Large Language Models via Complexity Classes","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.14890","snapshot_observed_at":"2026-08-07T04:23:50.574907Z","title":"Nphardeval: Dynamic benchmark on reasoning ability of large language models via complexity classes.arXiv preprint arXiv:2312.14890, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.574907Z"},"links":{"cited_paper":"/paper/2312.14890","citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:f17866fe6c032a09ef3cc0da91e617f5128d9e749768d2380e6f8a971ea3f52b","observation_id":"54c5a251-52c7-496e-97db-ec3ed7ebaf2d","resolution":{"observed_at":"2026-08-07T04:23:50.574907Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05530","last_updated":"2024-12-16T17:39:39Z","snapshot_observed_at":"2026-07-06T17:41:42.995949Z","submitted_at":"2024-03-08T18:54:20Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05530","snapshot_observed_at":"2026-08-07T04:23:50.577993Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.577993Z"},"links":{"cited_paper":"/paper/2403.05530","citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:d5be56b860764f79d5ca2c67bd6a38e69d8f4759b747c206b1047ee88231f0b4","observation_id":"a424da53-d40a-46f5-af3b-ef1a0875be4c","resolution":{"observed_at":"2026-08-07T04:23:50.577993Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-07T04:23:50.580934Z","title":"The llama 3 herd of models.arXiv preprint arXiv:2407.21783, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.580934Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:902689bc77c5a8683e252d86f4eeffda8b033f93c98fa72a46e8799073edaca9","observation_id":"bc03159f-8cdf-4858-901d-19719e22bfa7","resolution":{"observed_at":"2026-08-07T04:23:50.580934Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-07T04:23:50.584251Z","title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning.arXiv preprint arXiv:2501.12948, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.584251Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:7d8cd64489dea77b2e56da51bbe4c4787a11c98a7487e37430738234aa0c65dd","observation_id":"7cbf9a37-12b5-4b76-aecd-cc4a7a198749","resolution":{"observed_at":"2026-08-07T04:23:50.584251Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.02018","last_updated":"2025-05-04T07:48:36Z","snapshot_observed_at":"2026-08-06T22:24:55.181741Z","submitted_at":"2025-05-04T07:48:36Z","title":"R-Bench: Graduate-level Multi-disciplinary Benchmarks for LLM & MLLM Complex Reasoning Evaluation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.02018","snapshot_observed_at":"2026-08-07T04:23:50.587173Z","title":"R-bench: Graduate-level multi-disciplinary benchmarks for llm & mllm complex reasoning evaluation.arXiv preprint arXiv:2505.02018, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.587173Z"},"links":{"cited_paper":"/paper/2505.02018","citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:9565efd27738e2ab535a4f2ec7310a9b04270525d45beda7ec6842c48fc1074c","observation_id":"4380e4f1-ea7e-40b0-b55e-c85b8b215dc0","resolution":{"observed_at":"2026-08-07T04:23:50.587173Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2009.03300","last_updated":"2021-01-12T18:57:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-09-07T17:59:25Z","title":"Measuring Massive Multitask Language Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.03300","snapshot_observed_at":"2026-08-07T04:23:50.590207Z","title":"Measuring massive multitask language understanding.arXiv preprint arXiv:2009.03300, 2020","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.590207Z"},"links":{"cited_paper":"/paper/2009.03300","citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:af7e9629870284e301f4c896ef27a095f42365a23433aac17ad5e30714d47942","observation_id":"94684895-c743-40d3-87f1-6e6be6a6bbe9","resolution":{"observed_at":"2026-08-07T04:23:50.590207Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2103.03874","last_updated":"2021-11-08T21:30:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-03-05T18:59:39Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.03874","snapshot_observed_at":"2026-08-07T04:23:50.593077Z","title":"Measuring mathematical problem solving with the math dataset","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.593077Z"},"links":{"cited_paper":"/paper/2103.03874","citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:6becde493d103f558313673d9cefe9a912b372c9055a93df2f70d328425b0d4d","observation_id":"5b1a572f-7661-4758-a832-1cfbdf5d1d36","resolution":{"observed_at":"2026-08-07T04:23:50.593077Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:23:50.596213Z","title":"Mlagentbench: Evaluating language agents on machine learning experimentation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.596213Z"},"links":{"citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:cbd5cf1c0c9a4cf4ba4acecaaa40bab85a06fde75642d4b17d0ca8b28449943e","observation_id":"59c6256c-9287-43d1-96eb-183abc89676b","resolution":{"observed_at":"2026-08-07T04:23:50.596213Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:23:50.599339Z","title":"Aide: Ai-driven exploration in the space of code, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.599339Z"},"links":{"citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:ee114aefece04efbfb136a6496e3564df2ab3fca3ceba3486c541624bb3c0cc3","observation_id":"141b4162-76ee-4f17-8f17-b187f4a13aa0","resolution":{"observed_at":"2026-08-07T04:23:50.599339Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:23:50.602556Z","title":"Let’s verify step by step","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.602556Z"},"links":{"citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:3c39eaae863da5d8dddfa6a34ad86cc89ac5405016a6af5ab798870aaacaa3f3","observation_id":"316f20a2-a8c8-4f4d-923d-612cab304516","resolution":{"observed_at":"2026-08-07T04:23:50.602556Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:23:51.234176Z","title":"Truthfulqa: Measuring how models mimic human falsehoods","venue":null,"work_id":"a811b3d6-73dc-4a5c-ae28-5d249d56640c","year":2022},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.605362Z"},"links":{"citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:23e6a205b154f59f4763c35c45cceaa27609edd0d4cc39a653f25fe5dbfa53b2","observation_id":"4ae518c8-e72e-4df0-8705-bd7480778af1","resolution":{"observed_at":"2026-08-07T04:23:51.237836Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:23:51.224012Z","title":"Criticbench: Benchmarking llms for critique-correct reasoning, 2024","venue":null,"work_id":"0deaaeff-4d74-43d2-874b-ff5b8d2fd455","year":2024},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.608019Z"},"links":{"citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:d3202107fc80e6cfddce3b2b15720f52e413a5a70ff1a1676a95ef9bbf0466bf","observation_id":"c22a6d6e-7175-4207-896c-7be959e993f2","resolution":{"observed_at":"2026-08-07T04:23:51.227353Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.03688","last_updated":"2025-10-04T03:54:18Z","snapshot_observed_at":"2026-08-06T20:36:41.418114Z","submitted_at":"2023-08-07T16:08:11Z","title":"AgentBench: Evaluating LLMs as Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.03688","snapshot_observed_at":"2026-08-07T04:23:50.610680Z","title":"Agentbench: Evaluating llms as agents.arXiv preprint arXiv:2308.03688, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.610680Z"},"links":{"cited_paper":"/paper/2308.03688","citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:4eb0faf3c2701bade9623dce452864051147ccea39b18f507061acbae0d314b4","observation_id":"cbe914bb-a89b-488e-9a4e-23270b03daf5","resolution":{"observed_at":"2026-08-07T04:23:50.610680Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.17651","last_updated":"2023-05-25T19:13:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-30T18:30:01Z","title":"Self-Refine: Iterative Refinement with Self-Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.17651","snapshot_observed_at":"2026-08-07T04:23:50.613897Z","title":"Self-refine: Iterative refinement with self-feedback.arXiv preprint arXiv:2303.17651, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.613897Z"},"links":{"cited_paper":"/paper/2303.17651","citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:8bcd9a60617f8164f050212bbf9b610fa45bce32220e09a64c3045a55ea828bb","observation_id":"6fbbee3c-4310-4576-b5dd-86a596378fa4","resolution":{"observed_at":"2026-08-07T04:23:50.613897Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-07T04:23:50.616857Z","title":"Mle-bench: Evaluating machine learning agents on real-world machine learning engineering tasks.arXiv preprint arXiv:2410.07095, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.616857Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:c269ae78e8f1ba5c80be0e285089dcf9a6c5d531a04dc54b2f081ef2cb9026ea","observation_id":"b37b2185-8b21-4127-a86d-11fa9a8c23d5","resolution":{"observed_at":"2026-08-07T04:23:50.616857Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:23:51.214259Z","title":"Openai o1 system card","venue":null,"work_id":"34eac33f-c25c-4d8d-9bb6-aff906930aaf","year":2025},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.619574Z"},"links":{"citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:dbc8280307cd7f1f2497ed9a191165c493823e07fa223db326b9af574d11d8b6","observation_id":"3ad8f589-dcb4-4b9e-bcbd-43b850d2aae7","resolution":{"observed_at":"2026-08-07T04:23:51.217387Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:23:51.204321Z","title":"Toolbench: An open platform for training, serving, and evaluating large language models as tool agents.https://github.com/OpenBMB/ToolBench, 2023","venue":null,"work_id":"6ac7061f-6ceb-4989-bee2-21835c658523","year":2023},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.622398Z"},"links":{"citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:38c848cf56f4c66b457baedc3bf1d8b069e13b2a103878d7a24d5370c5d6d3e4","observation_id":"70add6e7-969c-42c6-85f3-69a81e02bb88","resolution":{"observed_at":"2026-08-07T04:23:51.207456Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:23:50.625582Z","title":"Training language models to follow instructions with human feedback.Advances in neural information processing systems, 35:27730–27744, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.625582Z"},"links":{"citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:aaf0e39006307b33e55f5ffc011fab2efa61905d09aef061bd3189280604a762","observation_id":"88b56bbd-2842-4e35-9c4a-b5e20125ad00","resolution":{"observed_at":"2026-08-07T04:23:50.625582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.09014","last_updated":"2023-03-16T01:04:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-16T01:04:45Z","title":"ART: Automatic multi-step reasoning and tool-use for large language models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.09014","snapshot_observed_at":"2026-08-07T04:23:50.628403Z","title":"Art: Automatic multi-step reasoning and tool-use for language models.arXiv preprint arXiv:2303.09014, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.628403Z"},"links":{"cited_paper":"/paper/2303.09014","citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:8c304add29a56bda87fa3267ddbf0d2544fbb5b79ad91b71331c82f69d74f8d7","observation_id":"aa9aceab-8e31-46f2-a87d-48e6db534039","resolution":{"observed_at":"2026-08-07T04:23:50.628403Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.04761","last_updated":"2023-02-09T16:49:57Z","snapshot_observed_at":"2026-07-06T14:50:07.491434Z","submitted_at":"2023-02-09T16:49:57Z","title":"Toolformer: Language Models Can Teach Themselves to Use Tools","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.04761","snapshot_observed_at":"2026-08-07T04:23:50.631361Z","title":"Toolformer: Language models can teach themselves to use tools.arXiv preprint arXiv:2302.04761, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.631361Z"},"links":{"cited_paper":"/paper/2302.04761","citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:8ad4c64f4bf7a1440631701e76da9c1620265c915a7d8f162038ceb1a1ff92fa","observation_id":"04db7ec6-6883-4523-be2d-152247168773","resolution":{"observed_at":"2026-08-07T04:23:50.631361Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.11366","last_updated":"2023-10-10T05:21:45Z","snapshot_observed_at":"2026-07-06T15:05:53.556198Z","submitted_at":"2023-03-20T18:08:50Z","title":"Reflexion: Language Agents with Verbal Reinforcement Learning","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.11366","snapshot_observed_at":"2026-08-07T04:23:50.634456Z","title":"Reflexion: Language agents with verbal reinforcement learning.arXiv preprint arXiv:2303.11366, 14, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.634456Z"},"links":{"cited_paper":"/paper/2303.11366","citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:68b5016b382a02668ccf6079cb9b426bb1ed3257551cf3cc0a1402e6d071cb81","observation_id":"b5a94c9a-ccbe-4db5-ba5d-22ef39cabf80","resolution":{"observed_at":"2026-08-07T04:23:50.634456Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.03768","last_updated":"2021-03-14T22:44:38Z","snapshot_observed_at":"2026-07-06T10:02:33.297722Z","submitted_at":"2020-10-08T05:13:36Z","title":"ALFWorld: Aligning Text and Embodied Environments for Interactive Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.03768","snapshot_observed_at":"2026-08-07T04:23:50.637541Z","title":"Alfworld: Aligning text and embodied environments for interactive learning.arXiv preprint arXiv:2010.03768, 2020","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.637541Z"},"links":{"cited_paper":"/paper/2010.03768","citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:63df654f6c4c64a00c00cef884ee8e8a0a46c13893ed2f8d3dba0077d7f02df4","observation_id":"cde1fcaf-a9ed-4c4d-b2fc-0222ca212c1b","resolution":{"observed_at":"2026-08-07T04:23:50.637541Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.04615","last_updated":"2023-06-12T17:51:15Z","snapshot_observed_at":"2026-07-06T13:19:12.109592Z","submitted_at":"2022-06-09T17:05:34Z","title":"Beyond the Imitation Game: Quantifying and extrapolating the capabilities of language models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.04615","snapshot_observed_at":"2026-08-07T04:23:50.640693Z","title":"Beyond the imitation game: Quantifying and extrapolating the capabilities of language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.640693Z"},"links":{"cited_paper":"/paper/2206.04615","citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:6d2dcab5e5c4fa6a92447962ba568cc575c684a1768b8dff4216405d1c6952d0","observation_id":"5c483474-2b68-44a9-af0b-9fb159ef37a7","resolution":{"observed_at":"2026-08-07T04:23:50.640693Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:23:51.188946Z","title":"Commonsenseqa: A question answering challenge targeting commonsense knowledge","venue":null,"work_id":"848f9657-d9e7-4464-a760-c41a63c93757","year":2019},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.643779Z"},"links":{"citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:72b62158af08090e0088ff3a9dd97e43478879e26eae00668b8a4e2fdb282767","observation_id":"1cedfafc-06fb-410e-80c2-4218bf49f1f6","resolution":{"observed_at":"2026-08-07T04:23:51.192086Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-02T11:57:18.735747Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-07T04:23:50.646976Z","title":"Llama 2: Open foundation and fine-tuned chat models.arXiv preprint arXiv:2307.09288, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.646976Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:e3540a2e3f7c1b9f381bcceb79cbfceeb561aba1ec52cb71ab3838711312d987","observation_id":"e0742482-4557-4ffc-802a-7fe44fd554b8","resolution":{"observed_at":"2026-08-07T04:23:50.646976Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:23:50.650424Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.650424Z"},"links":{"citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:15210f4ce1d53d24b22fa529d553c19eb51ed7501d454083e7358ea70ef9b2bd","observation_id":"0ee32718-1a8f-4d23-80ac-5bf23969389c","resolution":{"observed_at":"2026-08-07T04:23:50.650424Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:23:51.172438Z","title":"Superglue: A stickier benchmark for general-purpose language understanding systems","venue":null,"work_id":"0b2cbea0-cafe-43d4-98e1-813d987fb6f7","year":2019},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.653556Z"},"links":{"citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:e5ac73c47afa9dfa0bd2ff2fb5128ef8a1413f6e237a2431c1489c8178a0af40","observation_id":"77c7a4e8-b25b-4364-bbba-1c434b0551d9","resolution":{"observed_at":"2026-08-07T04:23:51.175816Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:23:51.161423Z","title":"Glue: A multi-task benchmark and analysis platform for natural language understanding","venue":null,"work_id":"22e1b026-53ee-455f-82d4-d04d56ac0a56","year":2018},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.656382Z"},"links":{"citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:1c11e565bcb4e7034820ee1c2fd929d0cb48643d693fb84231ab312d938fdc4b","observation_id":"30ab32ec-0e81-4648-bd65-b0f1709bc8ab","resolution":{"observed_at":"2026-08-07T04:23:51.165231Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:23:50.659110Z","title":"Chain-of-thought prompting elicits reasoning in large language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.659110Z"},"links":{"citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:b9854e7bc9c1c09094fa93a5162476e2b6930418906fadf9f97de179318fe9dc","observation_id":"a3adac98-9a74-4d92-a4ca-8cc85b9b86ed","resolution":{"observed_at":"2026-08-07T04:23:50.659110Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:23:51.142380Z","title":"Are large language models really good logical reasoners? a comprehensive evaluation and beyond.IEEE Transactions on Knowledge and Data Engineering, 2025","venue":null,"work_id":"c1afb974-aedf-4ac0-98c8-ed9a6d7ed4f1","year":2025},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.661863Z"},"links":{"citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:30d06649c09b68e06bcb6e17a37e27065c1f5c8de131411d34d8110b812c2bec","observation_id":"d5d9f157-174c-4510-85da-0e779b3a2316","resolution":{"observed_at":"2026-08-07T04:23:51.147672Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14898","last_updated":"2023-10-30T17:52:18Z","snapshot_observed_at":"2026-07-06T15:46:55.111790Z","submitted_at":"2023-06-26T17:59:50Z","title":"InterCode: Standardizing and Benchmarking Interactive Coding with Execution Feedback","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.14898","snapshot_observed_at":"2026-08-07T04:23:50.664728Z","title":"Intercode: Standardizing and benchmarking interactive coding with execution feedback.arXiv preprint arXiv:2306.14898, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.664728Z"},"links":{"cited_paper":"/paper/2306.14898","citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:8fc024dabd19ddcc6f66fa3ce61204ea41b0735018d9b8b92207d582e80289d8","observation_id":"fe4b2975-5b95-4c5c-9a92-7fd4b40436d2","resolution":{"observed_at":"2026-08-07T04:23:50.664728Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2207.01206","last_updated":"2023-02-08T01:39:30Z","snapshot_observed_at":"2026-07-06T13:27:22.300465Z","submitted_at":"2022-07-04T05:30:22Z","title":"WebShop: Towards Scalable Real-World Web Interaction with Grounded Language Agents","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2207.01206","snapshot_observed_at":"2026-08-07T04:23:50.667812Z","title":"Webshop: Towards scalable real-world web interaction with grounded language agents.arXiv preprint arXiv:2207.01206, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.667812Z"},"links":{"cited_paper":"/paper/2207.01206","citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:095a5c37458918d4f037073ce67a1af6b1755dbef02a28cd98486a969803832e","observation_id":"651be0e9-89ba-4c4c-92e1-47304ff082b0","resolution":{"observed_at":"2026-08-07T04:23:50.667812Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.10601","last_updated":"2023-12-03T22:50:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-17T23:16:17Z","title":"Tree of Thoughts: Deliberate Problem Solving with Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.10601","snapshot_observed_at":"2026-08-07T04:23:50.671193Z","title":"Tree of thoughts: Deliberate problem solving with large language models.arXiv preprint arXiv:2305.10601, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.671193Z"},"links":{"cited_paper":"/paper/2305.10601","citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:591dac928712e823330e2e52446019b226d9c0f220c655606dceef65189816ea","observation_id":"fea9e1be-8724-4124-8197-19d1b07a18d8","resolution":{"observed_at":"2026-08-07T04:23:50.671193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.03629","last_updated":"2023-03-10T01:00:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-10-06T01:00:32Z","title":"ReAct: Synergizing Reasoning and Acting in Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.03629","snapshot_observed_at":"2026-08-07T04:23:50.674463Z","title":"React: Synergizing reasoning and acting in language models.arXiv preprint arXiv:2210.03629, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.674463Z"},"links":{"cited_paper":"/paper/2210.03629","citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:c92bfb0242311168ee34248a7e4c44059dd7e2858750817ce68d42495cdcfe0a","observation_id":"08e606c5-141b-4ebd-bbb3-d0b2510a6b24","resolution":{"observed_at":"2026-08-07T04:23:50.674463Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:23:50.677471Z","title":"Hellaswag: Can a machine really finish your sentence? InProceedings of the 57th Annual Meeting of the Association for Computational Linguistics, pages 4791–4800, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.677471Z"},"links":{"citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:fb51e58e7099277a1c727d789728ef943567f305a60434b863f2aed0900b9639","observation_id":"33d97532-bf7c-44fe-8194-c5efb9b6b978","resolution":{"observed_at":"2026-08-07T04:23:50.677471Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:23:50.680320Z","title":"Iolbench: Benchmarking llms on linguistic reasoning.arXiv preprint arXiv:2501.04249, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.680320Z"},"links":{"citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:c3ed1b93894c8849b2887c439c703cef05ea5b3b56816ff812db7d050fe95b12","observation_id":"88981e13-795f-4a16-931d-086f52242ac5","resolution":{"observed_at":"2026-08-07T04:23:50.680320Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.13854","last_updated":"2024-04-16T15:13:18Z","snapshot_observed_at":"2026-08-06T12:47:01.809185Z","submitted_at":"2023-07-25T22:59:32Z","title":"WebArena: A Realistic Web Environment for Building Autonomous Agents","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.13854","snapshot_observed_at":"2026-08-07T04:23:50.683069Z","title":"Webarena: A realistic web environment for building autonomous agents","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.683069Z"},"links":{"cited_paper":"/paper/2307.13854","citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:9d91a03fa6596442eda13c19f14b1336b641e10a02ae81645fcdb707fa98e1a6","observation_id":"50a6d4b1-5579-4ce8-ae5a-68fe93b66e54","resolution":{"observed_at":"2026-08-07T04:23:50.683069Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-07T04:16:34.945293Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems"},"reference_resolution":{"displayed":45,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":37,"verified_exact":0,"verified_fuzzy":8},"total_outbound_references":45},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 45 of 45 outbound references and 8 inbound Pith citation observations for arXiv:2506.10764."}