{"as_of":"2026-08-08T21:56:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:31a2bdc81b1b464d85715e6bc38cb3edcbf4d389d23b3ef1b90b9906aa1f7b37","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":20,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":20,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":20,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":20,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T04:23:50.664728Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-06-29T11:23:20.895534Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2306.14898","last_updated":"2023-10-30T17:52:18Z","snapshot_observed_at":"2026-07-06T15:46:55.111790Z","submitted_at":"2023-06-26T17:59:50Z","title":"InterCode: Standardizing and Benchmarking Interactive Coding with Execution Feedback","version":3},"cited_work":{"arxiv_id":"2306.14898","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.14898","snapshot_observed_at":"2026-06-29T11:23:20.895534Z","title":"Intercode: Standardizing and benchmarking interactive coding with execution feedback","venue":null,"work_id":"1c47eed5-7e6c-49c9-9060-401622d6c1e4","year":2023},"citing_paper":{"arxiv_id":"2404.07972","last_updated":"2024-05-30T08:55:12Z","snapshot_observed_at":"2026-08-07T10:10:50.201348Z","submitted_at":"2024-04-11T17:56:05Z","title":"OSWorld: Benchmarking Multimodal Agents for Open-Ended Tasks in Real Computer Environments","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-13T01:19:32.406859Z"},"links":{"cited_paper":"/paper/2306.14898","citing_paper":"/paper/2404.07972"},"observation_digest":"sha256:22b99a16d317988743bd02696b8ec2b4678c2456ac8dab2ee83ce8ab56668f2a","observation_id":"4e88b8a0-fefd-425c-8f95-4aee49b6af53","resolution":{"observed_at":"2026-05-13T01:19:32.507302Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14898","last_updated":"2023-10-30T17:52:18Z","snapshot_observed_at":"2026-07-06T15:46:55.111790Z","submitted_at":"2023-06-26T17:59:50Z","title":"InterCode: Standardizing and Benchmarking Interactive Coding with Execution Feedback","version":3},"cited_work":{"arxiv_id":"2306.14898","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.14898","snapshot_observed_at":"2026-06-29T11:23:20.895534Z","title":"Intercode: Standardizing and benchmarking interactive coding with execution feedback","venue":null,"work_id":"1c47eed5-7e6c-49c9-9060-401622d6c1e4","year":2023},"citing_paper":{"arxiv_id":"2408.00118","last_updated":"2024-10-02T15:22:49Z","snapshot_observed_at":"2026-08-02T16:20:09.773989Z","submitted_at":"2024-07-31T19:13:07Z","title":"Gemma 2: Improving Open Language Models at a Practical Size","version":3},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-05-10T12:11:16.326752Z"},"links":{"cited_paper":"/paper/2306.14898","citing_paper":"/paper/2408.00118"},"observation_digest":"sha256:f435103e160c0d739b401d8ef4444f4b3341ce71a33fc6fd7498e9345e07b7bb","observation_id":"c7fad0a8-726e-4417-be4f-a7eba076ae48","resolution":{"observed_at":"2026-05-10T12:11:16.397576Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14898","last_updated":"2023-10-30T17:52:18Z","snapshot_observed_at":"2026-07-06T15:46:55.111790Z","submitted_at":"2023-06-26T17:59:50Z","title":"InterCode: Standardizing and Benchmarking Interactive Coding with Execution Feedback","version":3},"cited_work":{"arxiv_id":"2306.14898","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.14898","snapshot_observed_at":"2026-06-29T11:23:20.895534Z","title":"Intercode: Standardizing and benchmarking interactive coding with execution feedback","venue":null,"work_id":"1c47eed5-7e6c-49c9-9060-401622d6c1e4","year":2023},"citing_paper":{"arxiv_id":"2409.12917","last_updated":"2024-10-04T17:28:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-19T17:16:21Z","title":"Training Language Models to Self-Correct via Reinforcement Learning","version":2},"reference_index":147,"source":"arxiv_source","source_observed_at":"2026-05-17T12:04:10.210508Z"},"links":{"cited_paper":"/paper/2306.14898","citing_paper":"/paper/2409.12917"},"observation_digest":"sha256:2b70062de4be3bb3155dc43a1dacb42b7432e2ecce2a1f253c917e27183c6d8f","observation_id":"539ad4c5-a155-4bb9-aa70-e2f6f2e740f2","resolution":{"observed_at":"2026-05-17T12:04:10.693845Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14898","last_updated":"2023-10-30T17:52:18Z","snapshot_observed_at":"2026-07-06T15:46:55.111790Z","submitted_at":"2023-06-26T17:59:50Z","title":"InterCode: Standardizing and Benchmarking Interactive Coding with Execution Feedback","version":3},"cited_work":{"arxiv_id":"2306.14898","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.14898","snapshot_observed_at":"2026-06-29T11:23:20.895534Z","title":"Intercode: Standardizing and benchmarking interactive coding with execution feedback","venue":null,"work_id":"1c47eed5-7e6c-49c9-9060-401622d6c1e4","year":2023},"citing_paper":{"arxiv_id":"2410.23218","last_updated":"2024-10-30T17:10:19Z","snapshot_observed_at":"2026-08-05T06:30:40.974506Z","submitted_at":"2024-10-30T17:10:19Z","title":"OS-ATLAS: A Foundation Action Model for Generalist GUI Agents","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-05-13T09:29:27.173784Z"},"links":{"cited_paper":"/paper/2306.14898","citing_paper":"/paper/2410.23218"},"observation_digest":"sha256:99e31d807faa1d8917652d9b4f7c15fc475d011ba99ccc2fb04e34b777d72dc4","observation_id":"a05762d5-d970-4558-a807-414ce6e31ab7","resolution":{"observed_at":"2026-05-13T09:29:27.725374Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14898","last_updated":"2023-10-30T17:52:18Z","snapshot_observed_at":"2026-07-06T15:46:55.111790Z","submitted_at":"2023-06-26T17:59:50Z","title":"InterCode: Standardizing and Benchmarking Interactive Coding with Execution Feedback","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.14898","snapshot_observed_at":"2026-08-07T04:23:50.664728Z","title":"Intercode: Standardizing and benchmarking interactive coding with execution feedback.arXiv preprint arXiv:2306.14898, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-08T09:28:21.214766Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.664728Z"},"links":{"cited_paper":"/paper/2306.14898","citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:901e753a45c46fd73a1b094c16620531e39902bbe8b600e820aa1ec10cecdb8d","observation_id":"fe4b2975-5b95-4c5c-9a92-7fd4b40436d2","resolution":{"observed_at":"2026-08-07T04:23:50.664728Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14898","last_updated":"2023-10-30T17:52:18Z","snapshot_observed_at":"2026-07-06T15:46:55.111790Z","submitted_at":"2023-06-26T17:59:50Z","title":"InterCode: Standardizing and Benchmarking Interactive Coding with Execution Feedback","version":3},"cited_work":{"arxiv_id":"2306.14898","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.14898","snapshot_observed_at":"2026-06-29T11:23:20.895534Z","title":"Intercode: Standardizing and benchmarking interactive coding with execution feedback","venue":null,"work_id":"1c47eed5-7e6c-49c9-9060-401622d6c1e4","year":2023},"citing_paper":{"arxiv_id":"2507.06261","last_updated":"2025-12-19T14:25:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-07T17:36:04Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","version":6},"reference_index":91,"source":"arxiv_source","source_observed_at":"2026-05-19T05:48:02.828938Z"},"links":{"cited_paper":"/paper/2306.14898","citing_paper":"/paper/2507.06261"},"observation_digest":"sha256:029f983bd5377c3b63064efd011b75827c5aa7ac6f46d799de0422157286a808","observation_id":"dc6a2e0e-4271-4bd0-820f-bdcb8a7cdd84","resolution":{"observed_at":"2026-05-19T05:52:07.763204Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14898","last_updated":"2023-10-30T17:52:18Z","snapshot_observed_at":"2026-07-06T15:46:55.111790Z","submitted_at":"2023-06-26T17:59:50Z","title":"InterCode: Standardizing and Benchmarking Interactive Coding with Execution Feedback","version":3},"cited_work":{"arxiv_id":"2306.14898","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.14898","snapshot_observed_at":"2026-06-29T11:23:20.895534Z","title":"Intercode: Standardizing and benchmarking interactive coding with execution feedback","venue":null,"work_id":"1c47eed5-7e6c-49c9-9060-401622d6c1e4","year":2023},"citing_paper":{"arxiv_id":"2507.14201","last_updated":"2026-05-01T04:31:46Z","snapshot_observed_at":"2026-08-03T00:52:00.145580Z","submitted_at":"2025-07-14T17:06:26Z","title":"ExCyTIn-Bench: Evaluating LLM agents on Cyber Threat Investigation","version":3},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-19T04:37:33.942379Z"},"links":{"cited_paper":"/paper/2306.14898","citing_paper":"/paper/2507.14201"},"observation_digest":"sha256:6116004c47f03c56e564b09cb6d8667edae85514509cfc08ef4a0ba6de4edc8d","observation_id":"421b26c3-468b-4eec-bfc1-9e04165ceaef","resolution":{"observed_at":"2026-05-19T04:42:04.731715Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14898","last_updated":"2023-10-30T17:52:18Z","snapshot_observed_at":"2026-07-06T15:46:55.111790Z","submitted_at":"2023-06-26T17:59:50Z","title":"InterCode: Standardizing and Benchmarking Interactive Coding with Execution Feedback","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.14898","snapshot_observed_at":"2026-08-06T14:25:15.844022Z","title":"Intercode: Standardizing and benchmarking interactive coding with execution feedback, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.19399","last_updated":"2025-07-25T16:06:16Z","snapshot_observed_at":"2026-08-07T13:02:25.826266Z","submitted_at":"2025-07-25T16:06:16Z","title":"Running in CIRCLE? A Simple Benchmark for LLM Code Interpreter Security","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-06T14:25:15.844022Z"},"links":{"cited_paper":"/paper/2306.14898","citing_paper":"/paper/2507.19399"},"observation_digest":"sha256:6ca2ee852371293c5953278e077711990a826c552aafa7f231925d1df1d926f3","observation_id":"4f30a2d7-2449-48b5-a025-1fc9b2f9f650","resolution":{"observed_at":"2026-08-06T14:25:15.844022Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14898","last_updated":"2023-10-30T17:52:18Z","snapshot_observed_at":"2026-07-06T15:46:55.111790Z","submitted_at":"2023-06-26T17:59:50Z","title":"InterCode: Standardizing and Benchmarking Interactive Coding with Execution Feedback","version":3},"cited_work":{"arxiv_id":"2306.14898","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.14898","snapshot_observed_at":"2026-06-29T11:23:20.895534Z","title":"Intercode: Standardizing and benchmarking interactive coding with execution feedback","venue":null,"work_id":"1c47eed5-7e6c-49c9-9060-401622d6c1e4","year":2023},"citing_paper":{"arxiv_id":"2604.15136","last_updated":"2026-04-16T15:15:58Z","snapshot_observed_at":"2026-07-06T23:02:44.990811Z","submitted_at":"2026-04-16T15:15:58Z","title":"Feedback-Driven Execution for LLM-Based Binary Analysis","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-10T10:40:32.133423Z"},"links":{"cited_paper":"/paper/2306.14898","citing_paper":"/paper/2604.15136"},"observation_digest":"sha256:cff96c41da93ed1749983d3c21b57516a2c7b1235336844bf40cdf75290f7204","observation_id":"ff2ff9f9-012e-4dad-932e-cec5c038a972","resolution":{"observed_at":"2026-05-10T10:44:37.936889Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14898","last_updated":"2023-10-30T17:52:18Z","snapshot_observed_at":"2026-07-06T15:46:55.111790Z","submitted_at":"2023-06-26T17:59:50Z","title":"InterCode: Standardizing and Benchmarking Interactive Coding with Execution Feedback","version":3},"cited_work":{"arxiv_id":"2306.14898","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.14898","snapshot_observed_at":"2026-06-29T11:23:20.895534Z","title":"Intercode: Standardizing and benchmarking interactive coding with execution feedback","venue":null,"work_id":"1c47eed5-7e6c-49c9-9060-401622d6c1e4","year":2023},"citing_paper":{"arxiv_id":"2604.18718","last_updated":"2026-04-20T18:17:51Z","snapshot_observed_at":"2026-08-05T14:04:49.062431Z","submitted_at":"2026-04-20T18:17:51Z","title":"Towards Optimal Agentic Architectures for Offensive Security Tasks","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-10T04:02:04.359269Z"},"links":{"cited_paper":"/paper/2306.14898","citing_paper":"/paper/2604.18718"},"observation_digest":"sha256:bb6687adcc9c330d3369c3f058ca61d4d0a5dac336ddecaa4b2c1cfe3d965b63","observation_id":"27b5ac5f-7a8c-49a3-a988-b6d093ff4ecb","resolution":{"observed_at":"2026-05-11T12:16:03.139155Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14898","last_updated":"2023-10-30T17:52:18Z","snapshot_observed_at":"2026-07-06T15:46:55.111790Z","submitted_at":"2023-06-26T17:59:50Z","title":"InterCode: Standardizing and Benchmarking Interactive Coding with Execution Feedback","version":3},"cited_work":{"arxiv_id":"2306.14898","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.14898","snapshot_observed_at":"2026-06-29T11:23:20.895534Z","title":"Intercode: Standardizing and benchmarking interactive coding with execution feedback","venue":null,"work_id":"1c47eed5-7e6c-49c9-9060-401622d6c1e4","year":2023},"citing_paper":{"arxiv_id":"2604.20148","last_updated":"2026-04-22T03:25:17Z","snapshot_observed_at":"2026-07-06T23:06:40.154278Z","submitted_at":"2026-04-22T03:25:17Z","title":"Meta-Tool: Efficient Few-Shot Tool Adaptation for Small Language Models","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-05-10T00:50:27.257888Z"},"links":{"cited_paper":"/paper/2306.14898","citing_paper":"/paper/2604.20148"},"observation_digest":"sha256:31ae45e47e4ed945a289e9d12d6196277b9826632501e71cbdff75c98bf15853","observation_id":"4842b0c6-41f6-44f9-9f67-e5f2811cbac0","resolution":{"observed_at":"2026-05-10T00:54:48.809796Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14898","last_updated":"2023-10-30T17:52:18Z","snapshot_observed_at":"2026-07-06T15:46:55.111790Z","submitted_at":"2023-06-26T17:59:50Z","title":"InterCode: Standardizing and Benchmarking Interactive Coding with Execution Feedback","version":3},"cited_work":{"arxiv_id":"2306.14898","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.14898","snapshot_observed_at":"2026-06-29T11:23:20.895534Z","title":"Intercode: Standardizing and benchmarking interactive coding with execution feedback","venue":null,"work_id":"1c47eed5-7e6c-49c9-9060-401622d6c1e4","year":2023},"citing_paper":{"arxiv_id":"2604.25727","last_updated":"2026-04-28T14:53:59Z","snapshot_observed_at":"2026-08-02T05:44:39.873306Z","submitted_at":"2026-04-28T14:53:59Z","title":"Toward Scalable Terminal Task Synthesis via Skill Graphs","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-07T16:13:50.484665Z"},"links":{"cited_paper":"/paper/2306.14898","citing_paper":"/paper/2604.25727"},"observation_digest":"sha256:db2403922d3ab4e54d529f464e4507bfe3284b6418e628000a29d7c034f8315e","observation_id":"4759661e-1d2d-4fcd-b639-63c929da966d","resolution":{"observed_at":"2026-05-11T23:51:16.938897Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14898","last_updated":"2023-10-30T17:52:18Z","snapshot_observed_at":"2026-07-06T15:46:55.111790Z","submitted_at":"2023-06-26T17:59:50Z","title":"InterCode: Standardizing and Benchmarking Interactive Coding with Execution Feedback","version":3},"cited_work":{"arxiv_id":"2306.14898","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.14898","snapshot_observed_at":"2026-06-29T11:23:20.895534Z","title":"Intercode: Standardizing and benchmarking interactive coding with execution feedback","venue":null,"work_id":"1c47eed5-7e6c-49c9-9060-401622d6c1e4","year":2023},"citing_paper":{"arxiv_id":"2605.08904","last_updated":"2026-05-09T11:51:34Z","snapshot_observed_at":"2026-07-06T23:21:02.177557Z","submitted_at":"2026-05-09T11:51:34Z","title":"OPT-BENCH: Evaluating the Iterative Self-Optimization of LLM Agents in Large-Scale Search Spaces","version":1},"reference_index":141,"source":"arxiv_source","source_observed_at":"2026-05-12T02:57:15.521594Z"},"links":{"cited_paper":"/paper/2306.14898","citing_paper":"/paper/2605.08904"},"observation_digest":"sha256:a03dd455f698919b28d7aa9eedd4595ccfbb71deecaa243e1f98f484506dcb3f","observation_id":"c5a1fe64-f8c3-442e-bf6b-15e7ce40cc5b","resolution":{"observed_at":"2026-05-12T03:01:18.775762Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14898","last_updated":"2023-10-30T17:52:18Z","snapshot_observed_at":"2026-07-06T15:46:55.111790Z","submitted_at":"2023-06-26T17:59:50Z","title":"InterCode: Standardizing and Benchmarking Interactive Coding with Execution Feedback","version":3},"cited_work":{"arxiv_id":"2306.14898","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.14898","snapshot_observed_at":"2026-06-29T11:23:20.895534Z","title":"Intercode: Standardizing and benchmarking interactive coding with execution feedback","venue":null,"work_id":"1c47eed5-7e6c-49c9-9060-401622d6c1e4","year":2023},"citing_paper":{"arxiv_id":"2605.10597","last_updated":"2026-05-11T14:01:36Z","snapshot_observed_at":"2026-07-06T23:22:32.858399Z","submitted_at":"2026-05-11T14:01:36Z","title":"CrackMeBench: Binary Reverse Engineering for Agents","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-05-12T04:44:13.078520Z"},"links":{"cited_paper":"/paper/2306.14898","citing_paper":"/paper/2605.10597"},"observation_digest":"sha256:7682eabfca825f787eff6a308ec69a54e6e738e2547ee93ccb47672f2f263126","observation_id":"bedc2437-7737-4586-b6fb-561fb2ff7552","resolution":{"observed_at":"2026-05-12T05:56:43.456149Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14898","last_updated":"2023-10-30T17:52:18Z","snapshot_observed_at":"2026-07-06T15:46:55.111790Z","submitted_at":"2023-06-26T17:59:50Z","title":"InterCode: Standardizing and Benchmarking Interactive Coding with Execution Feedback","version":3},"cited_work":{"arxiv_id":"2306.14898","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.14898","snapshot_observed_at":"2026-06-29T11:23:20.895534Z","title":"Intercode: Standardizing and benchmarking interactive coding with execution feedback","venue":null,"work_id":"1c47eed5-7e6c-49c9-9060-401622d6c1e4","year":2023},"citing_paper":{"arxiv_id":"2605.18073","last_updated":"2026-05-18T08:55:30Z","snapshot_observed_at":"2026-07-06T23:28:59.503300Z","submitted_at":"2026-05-18T08:55:30Z","title":"A-ProS: Towards Reliable Autonomous Programming Through Multi-Model Feedback","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-20T09:22:06.285118Z"},"links":{"cited_paper":"/paper/2306.14898","citing_paper":"/paper/2605.18073"},"observation_digest":"sha256:79fdcf72fca773201dc7633bf3c0848cd8e99e306015d6db155fdcb1bd48a0c0","observation_id":"cfefc68d-e3a0-4e4b-8aa8-a6f859724145","resolution":{"observed_at":"2026-05-20T09:23:10.555911Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14898","last_updated":"2023-10-30T17:52:18Z","snapshot_observed_at":"2026-07-06T15:46:55.111790Z","submitted_at":"2023-06-26T17:59:50Z","title":"InterCode: Standardizing and Benchmarking Interactive Coding with Execution Feedback","version":3},"cited_work":{"arxiv_id":"2306.14898","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.14898","snapshot_observed_at":"2026-06-29T11:23:20.895534Z","title":"Intercode: Standardizing and benchmarking interactive coding with execution feedback","venue":null,"work_id":"1c47eed5-7e6c-49c9-9060-401622d6c1e4","year":2023},"citing_paper":{"arxiv_id":"2605.29115","last_updated":"2026-05-27T21:23:00Z","snapshot_observed_at":"2026-08-01T21:39:08.620495Z","submitted_at":"2026-05-27T21:23:00Z","title":"unix-ctf: Procedural Environments for Unix-Competence Reinforcement Learning","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-29T11:19:38.959705Z"},"links":{"cited_paper":"/paper/2306.14898","citing_paper":"/paper/2605.29115"},"observation_digest":"sha256:d812ca8ba9d383e46e7148abe50c2f7e8e1f84c8c654ed1cef0a8e39c97dc386","observation_id":"06905fc6-0fae-4c00-9649-0a9350f6218b","resolution":{"observed_at":"2026-06-29T11:23:20.897179Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14898","last_updated":"2023-10-30T17:52:18Z","snapshot_observed_at":"2026-07-06T15:46:55.111790Z","submitted_at":"2023-06-26T17:59:50Z","title":"InterCode: Standardizing and Benchmarking Interactive Coding with Execution Feedback","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.14898","snapshot_observed_at":"2026-08-02T06:45:15.404937Z","title":"InterCode: Standardizing and benchmarking interactive coding with execution feedback","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.12161","last_updated":"2026-08-02T15:57:32Z","snapshot_observed_at":"2026-08-07T15:45:17.594747Z","submitted_at":"2026-07-13T21:10:22Z","title":"Token Reduction Is Not Cost Reduction","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-02T06:45:15.404937Z"},"links":{"cited_paper":"/paper/2306.14898","citing_paper":"/paper/2607.12161"},"observation_digest":"sha256:f1c86f1ec1f0af99ad70ca9441c198a945c2fea00327a1ea2462b081f84cd5b4","observation_id":"cde89d7d-a132-4ca7-81a8-df4fc26dbda3","resolution":{"observed_at":"2026-08-02T06:45:15.404937Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14898","last_updated":"2023-10-30T17:52:18Z","snapshot_observed_at":"2026-07-06T15:46:55.111790Z","submitted_at":"2023-06-26T17:59:50Z","title":"InterCode: Standardizing and Benchmarking Interactive Coding with Execution Feedback","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.14898","snapshot_observed_at":"2026-08-04T04:23:32.473137Z","title":"InterCode: Standardizing and benchmarking interactive coding with execution feedback","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.12161","last_updated":"2026-08-02T15:57:32Z","snapshot_observed_at":"2026-08-07T15:45:17.594747Z","submitted_at":"2026-07-13T21:10:22Z","title":"Token Reduction Is Not Cost Reduction","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-04T04:23:32.473137Z"},"links":{"cited_paper":"/paper/2306.14898","citing_paper":"/paper/2607.12161"},"observation_digest":"sha256:07166af1824d96128cfe33127d9ac3c48216b436200275a7ef98fc34d40efeaf","observation_id":"b5f905ca-bc27-4185-9528-d5ddceb2e1d9","resolution":{"observed_at":"2026-08-04T04:23:32.473137Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14898","last_updated":"2023-10-30T17:52:18Z","snapshot_observed_at":"2026-07-06T15:46:55.111790Z","submitted_at":"2023-06-26T17:59:50Z","title":"InterCode: Standardizing and Benchmarking Interactive Coding with Execution Feedback","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.14898","snapshot_observed_at":"2026-08-01T02:31:26.014099Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.25425","last_updated":"2026-07-28T08:21:41Z","snapshot_observed_at":"2026-08-06T12:23:00.828390Z","submitted_at":"2026-07-28T08:21:41Z","title":"The Disruptive Impact of Large Language Models on Capture the Flag Competitions and the Path Toward Fair Play","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-01T02:31:26.014099Z"},"links":{"cited_paper":"/paper/2306.14898","citing_paper":"/paper/2607.25425"},"observation_digest":"sha256:d1caa1409fb6d55ff230f91b402d55f158ed6cf1a22dc00fea0470f7316d667a","observation_id":"523c8696-edfe-4ff8-9149-b511a4f7c8d4","resolution":{"observed_at":"2026-08-01T02:31:26.014099Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14898","last_updated":"2023-10-30T17:52:18Z","snapshot_observed_at":"2026-07-06T15:46:55.111790Z","submitted_at":"2023-06-26T17:59:50Z","title":"InterCode: Standardizing and Benchmarking Interactive Coding with Execution Feedback","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.14898","snapshot_observed_at":"2026-08-04T00:48:33.449476Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.00326","last_updated":"2026-07-31T22:26:45Z","snapshot_observed_at":"2026-08-06T23:12:39.568094Z","submitted_at":"2026-07-31T22:26:45Z","title":"Learning to Coordinate Symbolic Tools: LLM Agents for Verified Sum-of-Squares Certificates","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-04T00:48:33.449476Z"},"links":{"cited_paper":"/paper/2306.14898","citing_paper":"/paper/2608.00326"},"observation_digest":"sha256:210cf5db95d4be81a608c3832274fa4e198a8f035e20358c88f1338044834417","observation_id":"e1d6829c-1f02-40f2-920f-7b5caa53f733","resolution":{"observed_at":"2026-08-04T00:48:33.449476Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2306.14898/citation-record","integrity":"/paper/2306.14898/integrity","json":"/paper/2306.14898/citation-record.json","paper":"/paper/2306.14898"},"outbound":[],"paper":{"arxiv_id":"2306.14898","last_updated":"2023-10-30T17:52:18Z","latest_version":3,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T15:46:55.111790Z","submitted_at":"2023-06-26T17:59:50Z","title":"InterCode: Standardizing and Benchmarking Interactive Coding with Execution Feedback"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 20 inbound Pith citation observations for arXiv:2306.14898."}