{"as_of":"2026-08-05T20:15:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b0efdfbdcfab9cce6543ca0369b81135b983cb048e190e43173ab8c53753297a","coverage":[{"denominator":20,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":20,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-11T03:37:07.841385Z","state":"measured"},{"denominator":120,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":120,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":217,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-03T07:17:13.420563Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-03T07:17:13.420563Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2601.20789","last_updated":"2026-05-29T01:36:45Z","snapshot_observed_at":"2026-08-05T12:56:01.162423Z","submitted_at":"2026-01-28T17:27:08Z","title":"SERA: Soft-Verified Efficient Repository Agents","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-03T07:17:13.420563Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2601.20789"},"observation_digest":"sha256:42c395c0980f08b57a0cb25a2411d72995d87b1646bde7ee71c6c15dd2c10f8e","observation_id":"bc2c7d59-08bb-43f3-8808-9fbaa0217091","resolution":{"observed_at":"2026-08-03T07:17:13.420563Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2602.02276","last_updated":"2026-02-02T16:17:38Z","snapshot_observed_at":"2026-07-06T22:44:09.804048Z","submitted_at":"2026-02-02T16:17:38Z","title":"Kimi K2.5: Visual Agentic Intelligence","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-10T16:09:05.225767Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2602.02276"},"observation_digest":"sha256:7c9e5d860bff6805ad1faf562f216981fd20cf9c43090b83488b5c526b6f2d1e","observation_id":"3cadcd68-2d55-4d6e-ac8d-6a4aa8a99808","resolution":{"observed_at":"2026-05-11T03:37:08.148477Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2602.11224","last_updated":"2026-04-28T16:08:25Z","snapshot_observed_at":"2026-07-06T22:45:29.609382Z","submitted_at":"2026-02-11T13:31:18Z","title":"Agent-Diff: Benchmarking LLM Agents on Enterprise API Tasks via Code Execution with State-Diff-Based Evaluation","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-16T03:04:17.755968Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2602.11224"},"observation_digest":"sha256:bb818ec6782f31134c346764ec62634e7e4afad6a95cbfb4ef514a061cfc13b4","observation_id":"8d070eb9-7a96-4dbd-8588-290400dabc58","resolution":{"observed_at":"2026-05-16T03:07:11.985780Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2602.22480","last_updated":"2026-06-02T16:57:10Z","snapshot_observed_at":"2026-08-02T20:45:10.800498Z","submitted_at":"2026-02-25T23:40:22Z","title":"VeRO: A Harness for Agents to Optimize Agents","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-15T19:01:50.547095Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2602.22480"},"observation_digest":"sha256:5859be75b3b244be4fe7d0194cce72f824a52b9898918121fbee9da0b068fcce","observation_id":"c2502f82-f2f0-4f31-bce7-5ad70f6b8715","resolution":{"observed_at":"2026-05-15T19:06:30.926621Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-02T20:45:12.321842Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2602.22480","last_updated":"2026-06-02T16:57:10Z","snapshot_observed_at":"2026-08-02T20:45:10.800498Z","submitted_at":"2026-02-25T23:40:22Z","title":"VeRO: A Harness for Agents to Optimize Agents","version":4},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-02T20:45:12.321842Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2602.22480"},"observation_digest":"sha256:2aa941885cbd926cb6a8cb0bbe6169f1b8269f26daf36d1e1fe55d15273954f7","observation_id":"3f66d6e5-9cbe-41ef-9b2e-302297c6924d","resolution":{"observed_at":"2026-08-02T20:45:12.321842Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2603.04601","last_updated":"2026-05-13T23:00:10Z","snapshot_observed_at":"2026-07-06T22:47:55.281598Z","submitted_at":"2026-03-04T21:00:33Z","title":"Vibe Code Bench: Evaluating AI Models on End-to-End Web Application Development","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-15T15:59:29.910200Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2603.04601"},"observation_digest":"sha256:fc2a4656a494a335b3058fb0df55e3db58998a6da2e379befc58526ebcd96010","observation_id":"e43ae4b7-e82f-472f-af09-1d853d86e373","resolution":{"observed_at":"2026-05-15T16:00:10.128726Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-07-13T20:49:09.477849Z","title":"Terminal-bench: Benchmarking agents on hard, realistic tasks in command line interfaces.arXiv preprint arXiv:2601.11868,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.21489","last_updated":"2026-07-08T16:30:55Z","snapshot_observed_at":"2026-08-03T00:37:21.202769Z","submitted_at":"2026-03-23T02:26:35Z","title":"Effective Strategies for Asynchronous Software Engineering Agents","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-07-13T20:49:09.477849Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2603.21489"},"observation_digest":"sha256:b52924028be76511c6f8773d36a19cedea6b3ddd89aefcbbafe9f865b6cad028","observation_id":"c361e55c-5065-4bb8-bfcb-60f76da75665","resolution":{"observed_at":"2026-07-13T20:49:09.477849Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2603.24755","last_updated":"2026-05-07T20:39:58Z","snapshot_observed_at":"2026-08-02T11:51:00.424293Z","submitted_at":"2026-03-25T19:26:44Z","title":"SlopCodeBench: Benchmarking How Coding Agents Degrade Over Long-Horizon Iterative Tasks","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-15T00:15:40.442807Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2603.24755"},"observation_digest":"sha256:75c5a138fc47c81b1d3b3421185e6bdaf7d0fbf14d8293f1292738c245692961","observation_id":"50b1f0e6-6b95-4ac1-a3fe-226c9e4e69ea","resolution":{"observed_at":"2026-05-15T00:18:21.683940Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2603.28052","last_updated":"2026-03-30T05:33:50Z","snapshot_observed_at":"2026-07-06T22:51:00.579062Z","submitted_at":"2026-03-30T05:33:50Z","title":"Meta-Harness: End-to-End Optimization of Model Harnesses","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-13T16:15:57.877354Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2603.28052"},"observation_digest":"sha256:a14ed45d37e4d0ce6a08108840babf57d1c9e5ad6c88301f11c2b5200ce175a9","observation_id":"c9e2e6ae-129b-440b-8428-c68864acdeec","resolution":{"observed_at":"2026-05-13T16:15:58.227250Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-07-13T09:29:51.204678Z","title":"Aniello Panariello, Daniel Marczak, Simone Magistri, Angelo Porrello, Bartlomiej Twar- dowski, Andrew D","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2604.05333","last_updated":"2026-05-27T08:01:13Z","snapshot_observed_at":"2026-07-13T09:29:44.671622Z","submitted_at":"2026-04-07T02:09:11Z","title":"Graph-of-Skills: Dependency-Aware Structural Retrieval for Massive Agent Skills","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-13T09:29:51.204678Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2604.05333"},"observation_digest":"sha256:20c1458024f5735b2a1a0b6ed8eecabf055e55f4fc85978cf0ab94420e648bb8","observation_id":"60c625e6-06e4-4417-aa07-988588164588","resolution":{"observed_at":"2026-07-13T09:29:51.204678Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2604.05336","last_updated":"2026-07-02T22:12:42Z","snapshot_observed_at":"2026-07-13T09:28:12.657391Z","submitted_at":"2026-04-07T02:22:44Z","title":"TRACE: Capability-Targeted Agentic Training","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T20:12:06.752461Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2604.05336"},"observation_digest":"sha256:e816a52da5790b0c7d66cb2d792381596b570f3ad70cb5c2180719d692f73ee3","observation_id":"b16f0811-0ba8-4acc-89f3-22f44b099f5d","resolution":{"observed_at":"2026-05-11T03:37:08.148477Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2604.06111","last_updated":"2026-04-10T03:07:44Z","snapshot_observed_at":"2026-08-02T21:04:18.191344Z","submitted_at":"2026-04-07T17:21:28Z","title":"AgentCE-Bench: Agent Configurable Evaluation with Scalable Horizons and Controllable Difficulty under Lightweight Environments","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-10T19:07:46.077831Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2604.06111"},"observation_digest":"sha256:953d3c3c5072362c524c73ce50e08dae3f03e67edf8c208336b0723851c95b51","observation_id":"2c22284f-4079-4a85-aa42-a9581bc39753","resolution":{"observed_at":"2026-05-11T03:37:08.148477Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2604.06132","last_updated":"2026-05-07T14:06:23Z","snapshot_observed_at":"2026-07-06T22:54:40.349479Z","submitted_at":"2026-04-07T17:43:18Z","title":"Claw-Eval: Towards Trustworthy Evaluation of Autonomous Agents","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-10T18:59:59.383781Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2604.06132"},"observation_digest":"sha256:6d5821f1f4e76acf76c48a4a5c7fc9794f8d4ad0c466d41e89255f53d0597e5b","observation_id":"a383907e-af74-40bd-9195-d767b431b951","resolution":{"observed_at":"2026-05-11T03:37:08.148477Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2604.06742","last_updated":"2026-07-17T08:33:08Z","snapshot_observed_at":"2026-08-03T21:08:28.866048Z","submitted_at":"2026-04-08T07:09:10Z","title":"Evaluating LLM-Based 0-to-1 Software Generation in End-to-End CLI Tool Scenarios","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-10T18:35:03.746555Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2604.06742"},"observation_digest":"sha256:6810f9b857956b8f1caa8bc992f044946bb98c6f9d8f1830a10b1322d2922d84","observation_id":"9d40e8b6-cfd2-4416-a969-90a908bbc508","resolution":{"observed_at":"2026-05-11T03:37:08.148477Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-02T16:42:14.837132Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2604.06742","last_updated":"2026-07-17T08:33:08Z","snapshot_observed_at":"2026-08-03T21:08:28.866048Z","submitted_at":"2026-04-08T07:09:10Z","title":"Evaluating LLM-Based 0-to-1 Software Generation in End-to-End CLI Tool Scenarios","version":2},"reference_index":2026,"source":"pdf_text","source_observed_at":"2026-08-02T16:42:14.837132Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2604.06742"},"observation_digest":"sha256:c935362adcd4022a40772c1eb6155ba70377a9e2af1537f5b84add203140a37e","observation_id":"4818b396-6ccc-4d60-9eec-fce92bdba149","resolution":{"observed_at":"2026-08-02T16:42:14.837132Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2604.07236","last_updated":"2026-04-28T14:27:47Z","snapshot_observed_at":"2026-07-06T22:55:33.502756Z","submitted_at":"2026-04-08T16:02:04Z","title":"How Much Heavy Lifting Can an Agent Harness Do?: Measuring the LLM's Residual Role in a Planning Agent","version":4},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:18.728493Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2604.07236"},"observation_digest":"sha256:ba5d10a91808265b2b2f6e8eb37288b885061e10cc6d9bf7f7d6afcbed47efb5","observation_id":"ecd9eaa4-0a65-46c0-a623-229feafe242a","resolution":{"observed_at":"2026-05-11T03:37:08.148477Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2604.07355","last_updated":"2026-03-28T06:13:17Z","snapshot_observed_at":"2026-07-06T22:55:37.787732Z","submitted_at":"2026-03-28T06:13:17Z","title":"Prediction Arena: Benchmarking AI Models on Real-World Prediction Markets","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-05-14T22:56:58.482896Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2604.07355"},"observation_digest":"sha256:79d4825493b5183c5d7aad35d5e6299257889617147ce2b928457df30e53b53d","observation_id":"01afe455-1add-420e-9124-63c932f63770","resolution":{"observed_at":"2026-05-14T22:58:14.109358Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2604.09836","last_updated":"2026-04-16T21:01:25Z","snapshot_observed_at":"2026-07-06T22:58:38.807460Z","submitted_at":"2026-04-10T19:08:50Z","title":"COMPOSITE-Stem","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-05-10T17:29:15.163053Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2604.09836"},"observation_digest":"sha256:8ca2fe28ae2b826f3cb1e79b29b1c64e113d32803140ee1aba4288caf7aaaa55","observation_id":"f81e0b6f-b431-4ec2-808e-eeb124a87ee5","resolution":{"observed_at":"2026-05-11T06:41:52.868766Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2604.10866","last_updated":"2026-04-16T16:00:33Z","snapshot_observed_at":"2026-07-06T22:59:23.126963Z","submitted_at":"2026-04-13T00:27:32Z","title":"OccuBench: Evaluating AI Agents on Real-World Professional Tasks via Language Environment Simulation","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-10T16:43:27.037355Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2604.10866"},"observation_digest":"sha256:e313c4cea7a05a5016a4128e3a8d738e10d8b7e6ae7e2c1a6779c2c704c983fb","observation_id":"d1cb947c-175e-4db4-9988-9380e13e9586","resolution":{"observed_at":"2026-05-11T08:20:57.808333Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2604.11078","last_updated":"2026-04-13T06:57:33Z","snapshot_observed_at":"2026-07-06T22:59:32.051572Z","submitted_at":"2026-04-13T06:57:33Z","title":"From Context to Rules: Toward Unified Detection Rule Generation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-10T16:25:59.609473Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2604.11078"},"observation_digest":"sha256:b49861cb093e64eac0c141d94239977810a925191e478df190de35dfec07a652","observation_id":"31d7448c-c4a4-4761-b50f-089bd20710e4","resolution":{"observed_at":"2026-05-11T08:55:59.987338Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2604.11518","last_updated":"2026-04-13T14:21:44Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-13T14:21:44Z","title":"From Translation to Superset: Benchmark-Driven Evolution of a Production AI Agent from Rust to Python","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T15:50:05.278161Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2604.11518"},"observation_digest":"sha256:8dac3f0344eaf167aaa541eda52e994a4f0503fe9dfa78f7d5e87b2a86a7fe67","observation_id":"24ac4b31-03b1-4834-8f89-731d838722dc","resolution":{"observed_at":"2026-05-11T09:46:08.268286Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2604.12290","last_updated":"2026-04-27T16:57:20Z","snapshot_observed_at":"2026-08-05T14:13:55.019169Z","submitted_at":"2026-04-14T05:02:06Z","title":"Frontier-Eng: Benchmarking Self-Evolving Agents on Real-World Engineering Tasks with Generative Optimization","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-10T16:17:32.290531Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2604.12290"},"observation_digest":"sha256:91e20fce78b9d8e37962d16ed70ccf0da5ae8416f35a575f48c76d48a69f321c","observation_id":"b4eb09ff-74d5-4d1f-a6e4-066780d33ace","resolution":{"observed_at":"2026-05-11T09:05:57.732149Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2604.12890","last_updated":"2026-04-25T02:32:57Z","snapshot_observed_at":"2026-07-06T23:01:00.830207Z","submitted_at":"2026-04-14T15:40:28Z","title":"Towards Long-horizon Agentic Multimodal Search","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-10T15:40:32.137708Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2604.12890"},"observation_digest":"sha256:fddae377aaedc715bfc4fd1b09a87fbf5da3ed80f2834e7a988b9699207422cc","observation_id":"9b070afe-5685-4a0c-87fd-f39ebfa076c0","resolution":{"observed_at":"2026-05-11T10:06:00.175925Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2604.13151","last_updated":"2026-04-14T17:59:57Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T17:59:57Z","title":"Exploration and Exploitation Errors Are Measurable for Language Model Agents","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-10T15:00:24.785343Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2604.13151"},"observation_digest":"sha256:ce1db230f7bffade7ec0018581ee6aac4a4fb60385f1854cc5c5892052ceabca","observation_id":"342b5f46-0949-417f-b797-6b2d68b7ad42","resolution":{"observed_at":"2026-05-11T11:21:02.382824Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2604.13536","last_updated":"2026-04-16T05:31:19Z","snapshot_observed_at":"2026-07-06T23:01:32.542040Z","submitted_at":"2026-04-15T06:32:07Z","title":"Don't Let AI Agents YOLO Your Files: Shifting Information and Control to Filesystems for Agent Safety and Autonomy","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-10T12:25:50.595161Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2604.13536"},"observation_digest":"sha256:d322b1c918225112aa977757d7b8a39dd7707852904976c75a7a0d27c2ee944c","observation_id":"ce1cea9c-d44b-491e-86d8-291b39ac4e5b","resolution":{"observed_at":"2026-05-11T03:37:08.148477Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2604.14709","last_updated":"2026-05-05T12:56:34Z","snapshot_observed_at":"2026-07-06T23:02:27.141640Z","submitted_at":"2026-04-16T07:19:34Z","title":"HWE-Bench: Benchmarking LLM Agents on Real-World Hardware Bug Repair Tasks","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-10T11:27:11.257817Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2604.14709"},"observation_digest":"sha256:a8d53bc631a0a5b6fa5de62dad5151af930b40d72cf1864a62bce98c08a79e6d","observation_id":"d6db8f4c-b25d-40c8-9fc5-d247cdfbb397","resolution":{"observed_at":"2026-05-11T03:37:08.148477Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2604.16682","last_updated":"2026-04-17T20:39:25Z","snapshot_observed_at":"2026-07-06T23:03:57.199731Z","submitted_at":"2026-04-17T20:39:25Z","title":"KAIROS: Stateful, Context-Aware Power-Efficient Agentic Inference Serving","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-10T06:58:36.525442Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2604.16682"},"observation_digest":"sha256:c3ec175d348477db037e52b3a0aee7b803fcf36ab70e6ca498124dc27a5c2d1d","observation_id":"e22b2236-bded-4744-b39e-beffc944a32f","resolution":{"observed_at":"2026-05-11T03:37:08.148477Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2604.17180","last_updated":"2026-04-19T00:47:24Z","snapshot_observed_at":"2026-08-03T03:46:10.762878Z","submitted_at":"2026-04-19T00:47:24Z","title":"BranchBench: Aligning Database Branching with Agentic Demands","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-10T06:12:46.899179Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2604.17180"},"observation_digest":"sha256:11cb22e76e08a28724a4c77ef0a886357319159e7284f560a0ac76b0b20f95b5","observation_id":"ebf058fe-d98b-4a30-a5a3-ba495e4e3a9a","resolution":{"observed_at":"2026-05-11T03:37:08.148477Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2604.17308","last_updated":"2026-04-19T07:51:46Z","snapshot_observed_at":"2026-07-06T23:04:28.260463Z","submitted_at":"2026-04-19T07:51:46Z","title":"SkillFlow:Benchmarking Lifelong Skill Discovery and Evolution for Autonomous Agents","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-10T06:13:32.434201Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2604.17308"},"observation_digest":"sha256:ad30bd54afa30975bc5aaff982cd19a646cd65f85b28c4781895b192bc0358f4","observation_id":"222dc975-e5bb-4afa-974d-e8eea22f129d","resolution":{"observed_at":"2026-05-11T03:37:08.148477Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2604.17596","last_updated":"2026-04-19T20:04:02Z","snapshot_observed_at":"2026-07-06T23:04:41.564912Z","submitted_at":"2026-04-19T20:04:02Z","title":"Terminal Wrench: A Dataset of 331 Reward-Hackable Environments and 3,632 Exploit Trajectories","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T05:48:44.687520Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2604.17596"},"observation_digest":"sha256:873fd7f8cac7534f7f305823a6a549ccf459f805cbe0e24189444dfb6309f1ab","observation_id":"b3620eee-499e-42a7-a8df-cab7d7b76c02","resolution":{"observed_at":"2026-05-11T03:37:08.148477Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2604.18292","last_updated":"2026-04-20T14:01:10Z","snapshot_observed_at":"2026-07-06T23:05:13.178333Z","submitted_at":"2026-04-20T14:01:10Z","title":"Agent-World: Scaling Real-World Environment Synthesis for Evolving General Agent Intelligence","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-10T05:24:00.503836Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2604.18292"},"observation_digest":"sha256:dbede02c09187f0dee47b01d2fbb8429d1a9bf459307d4cea9fd00866fc9d13f","observation_id":"73a253dd-f2fd-451e-b667-631dd09377f5","resolution":{"observed_at":"2026-05-11T03:37:08.148477Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2604.23781","last_updated":"2026-05-05T15:32:46Z","snapshot_observed_at":"2026-07-06T23:09:58.193241Z","submitted_at":"2026-04-26T16:05:02Z","title":"ClawMark: A Living-World Benchmark for Multi-Turn, Multi-Day, Multimodal Coworker Agents","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-08T06:36:52.202847Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2604.23781"},"observation_digest":"sha256:41fb059956715b690fa787d9855390fd1a56abeb27a93c92d24339dd11399f64","observation_id":"a4525fe0-769f-415d-9c41-d2dbc0339314","resolution":{"observed_at":"2026-05-11T21:11:10.854057Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2604.25727","last_updated":"2026-04-28T14:53:59Z","snapshot_observed_at":"2026-08-02T05:44:39.873306Z","submitted_at":"2026-04-28T14:53:59Z","title":"Toward Scalable Terminal Task Synthesis via Skill Graphs","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-07T16:13:50.484665Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2604.25727"},"observation_digest":"sha256:ebeded8e67950fb1b497454b06ca29aa17ef730e996d8dbe9333d2e5b9dcdd0e","observation_id":"8c5651dc-ed4c-4bbb-8999-66f412d579d6","resolution":{"observed_at":"2026-05-11T23:51:16.809578Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2604.25850","last_updated":"2026-05-18T15:11:47Z","snapshot_observed_at":"2026-07-06T23:11:39.318906Z","submitted_at":"2026-04-28T16:55:02Z","title":"Agentic Harness Engineering: Observability-Driven Automatic Evolution of Coding-Agent Harnesses","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-07T16:18:46.514331Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2604.25850"},"observation_digest":"sha256:d748a80a7d6d8296c47abf85047506840dccd74876c2ac859b5c8ea55d097b93","observation_id":"96afacb2-6ed8-4f57-a9e4-1b1a3d06d911","resolution":{"observed_at":"2026-05-11T23:46:45.190352Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2604.25850","last_updated":"2026-05-18T15:11:47Z","snapshot_observed_at":"2026-07-06T23:11:39.318906Z","submitted_at":"2026-04-28T16:55:02Z","title":"Agentic Harness Engineering: Observability-Driven Automatic Evolution of Coding-Agent Harnesses","version":4},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-20T23:50:56.311572Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2604.25850"},"observation_digest":"sha256:e7c48744c4496bc689c79ddf67e73cb180299ccd1c4afc7f3c9a9e116f64950e","observation_id":"a92fa150-3bf8-4f25-b86a-c31343bdc321","resolution":{"observed_at":"2026-05-20T23:53:51.714728Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2604.26963","last_updated":"2026-04-14T05:15:28Z","snapshot_observed_at":"2026-08-01T00:07:50.693557Z","submitted_at":"2026-04-14T05:15:28Z","title":"MARS: Efficient, Adaptive Co-Scheduling for Heterogeneous Agentic Systems","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-10T14:30:56.899306Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2604.26963"},"observation_digest":"sha256:6317bf4201e14ad6f5a24989fab92584897c6c629071830d0b31d6ce8b9da546","observation_id":"0fac8e12-f199-403e-8617-2cb5f533c068","resolution":{"observed_at":"2026-05-11T03:37:08.148477Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2604.27351","last_updated":"2026-04-30T03:02:27Z","snapshot_observed_at":"2026-07-06T23:12:52.141592Z","submitted_at":"2026-04-30T03:02:27Z","title":"Heterogeneous Scientific Foundation Model Collaboration","version":1},"reference_index":107,"source":"pdf_text","source_observed_at":"2026-05-07T08:50:05.980191Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2604.27351"},"observation_digest":"sha256:3d89f1135a416cc7eab9ffa6fdf69e01a0995146776463220b0dbdabfd9c9298","observation_id":"2f684907-36ef-4aff-a4df-fb2158e37a91","resolution":{"observed_at":"2026-05-11T03:37:08.148477Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2604.28093","last_updated":"2026-04-30T16:37:37Z","snapshot_observed_at":"2026-08-05T05:41:38.682049Z","submitted_at":"2026-04-30T16:37:37Z","title":"What Makes a Good Terminal-Agent Benchmark Task: A Guideline for Adversarial, Difficult, and Legible Evaluation Design","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-07T05:10:58.065201Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2604.28093"},"observation_digest":"sha256:c4c8d86e6f2c71b2a7987b27955c71a675c0781831b37e8aa6e13e646624cc7a","observation_id":"1b7cf7a6-cd43-4d3a-949f-0897f251da6e","resolution":{"observed_at":"2026-05-12T10:36:30.975678Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2604.28138","last_updated":"2026-04-30T17:20:19Z","snapshot_observed_at":"2026-08-02T08:34:04.235617Z","submitted_at":"2026-04-30T17:20:19Z","title":"Crab: A Semantics-Aware Checkpoint/Restore Runtime for Agent Sandboxes","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-07T05:56:11.552190Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2604.28138"},"observation_digest":"sha256:d7d8b3951539c396fe7337ab12d28ffb1a960142ad1fc7656933735c7410cfa5","observation_id":"9546a39b-4954-431b-9a4a-4abc9fc99d66","resolution":{"observed_at":"2026-05-12T10:31:28.618912Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.00505","last_updated":"2026-05-17T17:18:53Z","snapshot_observed_at":"2026-07-06T23:13:57.367556Z","submitted_at":"2026-05-01T08:30:52Z","title":"LLM-Oriented Information Retrieval: A Denoising-First Perspective","version":1},"reference_index":126,"source":"pdf_text","source_observed_at":"2026-05-09T18:54:06.144968Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.00505"},"observation_digest":"sha256:e43c4157a14aac352a3c952c837e381b8f4297dd57bfa9e3f3973f816e48bebb","observation_id":"6bdf56a8-1aed-4386-ae4d-fb2556d38d18","resolution":{"observed_at":"2026-05-11T16:01:19.839855Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.00505","last_updated":"2026-05-17T17:18:53Z","snapshot_observed_at":"2026-07-06T23:13:57.367556Z","submitted_at":"2026-05-01T08:30:52Z","title":"LLM-Oriented Information Retrieval: A Denoising-First Perspective","version":2},"reference_index":129,"source":"pdf_text","source_observed_at":"2026-05-21T00:18:32.423103Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.00505"},"observation_digest":"sha256:474eceded70db0b7aadfe9b9eaa11d4335fe822b9faef458e3bebcba3a2593cd","observation_id":"1c02537c-702a-425d-9fc1-fb4fe28c57a1","resolution":{"observed_at":"2026-05-21T00:19:16.648530Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.06136","last_updated":"2026-05-07T12:35:27Z","snapshot_observed_at":"2026-07-06T23:18:41.400741Z","submitted_at":"2026-05-07T12:35:27Z","title":"BUILD-AND-FIND: An Effort-Aware Protocol for Evaluating Agent-Managed Codebases","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-08T08:57:41.818402Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.06136"},"observation_digest":"sha256:3c2994875f25c79e682bd6a56307300b33b67870af70300635586785a5da8ff1","observation_id":"961c9605-9309-40bd-9df3-2f434e68e2d8","resolution":{"observed_at":"2026-05-11T20:26:12.626586Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.06455","last_updated":"2026-05-07T15:49:48Z","snapshot_observed_at":"2026-08-05T02:36:24.838528Z","submitted_at":"2026-05-07T15:49:48Z","title":"PrefixGuard: From LLM-Agent Traces to Online Failure-Warning Monitors","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-08T09:44:55.593277Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.06455"},"observation_digest":"sha256:3b7f1ec214d21cd276672d15e495cdf0f4b7d95fa9521bdba92607bd730f4dfe","observation_id":"6c3523ec-d2b4-49d8-baf1-3470940b2b51","resolution":{"observed_at":"2026-05-11T20:16:10.832441Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.07073","last_updated":"2026-05-08T00:48:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-05-08T00:48:45Z","title":"TeamBench: Evaluating Agent Coordination under Enforced Role Separation","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-11T00:55:51.358828Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.07073"},"observation_digest":"sha256:dc54893e916452c440bf80598d4c79e1bcfca021289c03be803d51f2893c49c7","observation_id":"0ebd8258-2635-42b0-acfa-36a76a63ddb0","resolution":{"observed_at":"2026-05-11T05:00:55.485805Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.07161","last_updated":"2026-07-30T18:34:51Z","snapshot_observed_at":"2026-08-05T19:21:20.472460Z","submitted_at":"2026-05-08T02:47:07Z","title":"SREGym: A Live Benchmark for AI SRE Agents with High-Fidelity Failure Scenarios","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-11T02:20:32.528550Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.07161"},"observation_digest":"sha256:d1d366b7574d9554c9dfb1163a55adbc993609e9230fa6ac56bb4b34efbb1a88","observation_id":"bb9829ac-ef4d-47a2-961c-d4ea2a3aded4","resolution":{"observed_at":"2026-05-11T03:37:08.148477Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.07161","last_updated":"2026-07-30T18:34:51Z","snapshot_observed_at":"2026-08-05T19:21:20.472460Z","submitted_at":"2026-05-08T02:47:07Z","title":"SREGym: A Live Benchmark for AI SRE Agents with High-Fidelity Failure Scenarios","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-14T21:49:16.239350Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.07161"},"observation_digest":"sha256:a0e3fb6991d0cc465d5a5d41117a799faffde3081777edc6891916941f828854","observation_id":"61cd7e8a-a615-4554-b1d0-b48146f8d843","resolution":{"observed_at":"2026-05-14T21:49:29.207918Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-03T00:18:43.017866Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2605.07161","last_updated":"2026-07-30T18:34:51Z","snapshot_observed_at":"2026-08-05T19:21:20.472460Z","submitted_at":"2026-05-08T02:47:07Z","title":"SREGym: A Live Benchmark for AI SRE Agents with High-Fidelity Failure Scenarios","version":3},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-03T00:18:43.017866Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.07161"},"observation_digest":"sha256:9b8833d9a13eafbedff21a3387f8add1573d119e70543168b3fd07f64760a83a","observation_id":"decc0cb2-5d3a-4d38-bb4c-c0ce013a4088","resolution":{"observed_at":"2026-08-03T00:18:43.017866Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.08013","last_updated":"2026-05-08T17:02:31Z","snapshot_observed_at":"2026-07-06T23:20:19.821342Z","submitted_at":"2026-05-08T17:02:31Z","title":"Learning CLI Agents with Structured Action Credit under Selective Observation","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-11T02:59:26.100818Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.08013"},"observation_digest":"sha256:f9e99cbaf2dabfdb1726ef2aa9cc1c724fd94225780d246d4a5ece2bd090d5ed","observation_id":"9aebe349-74e9-4a5e-8932-3b0870b4ed46","resolution":{"observed_at":"2026-05-11T03:37:08.148477Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.08366","last_updated":"2026-05-08T18:21:44Z","snapshot_observed_at":"2026-07-06T23:20:38.385189Z","submitted_at":"2026-05-08T18:21:44Z","title":"SWE Atlas: Benchmarking Coding Agents Beyond Issue Resolution","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-12T02:28:07.557119Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.08366"},"observation_digest":"sha256:bfed4146ea6db0843a8c72f663df2093a3183d509cb80e3d5f4330e116d98fc7","observation_id":"9460b06a-2718-4a8d-824e-c7412ab0b561","resolution":{"observed_at":"2026-05-12T07:36:58.326900Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.08941","last_updated":"2026-05-09T13:21:51Z","snapshot_observed_at":"2026-07-31T16:01:12.967586Z","submitted_at":"2026-05-09T13:21:51Z","title":"MDGYM: Benchmarking AI Agents on Molecular Simulations","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-07-11T11:50:26.030339Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.08941"},"observation_digest":"sha256:2606f4bd16ed1235b06af2d1a9e682d33ccc1d1ad927591ee52cde89d42e260c","observation_id":"dcafdf38-1693-42ab-86cf-5468dfba4fc0","resolution":{"observed_at":"2026-05-12T03:16:18.565184Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.09252","last_updated":"2026-05-20T19:57:19Z","snapshot_observed_at":"2026-08-02T16:03:49.427921Z","submitted_at":"2026-05-10T01:37:40Z","title":"LLM Agents Already Know When to Call Tools -- Even Without Reasoning","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-05-12T02:44:00.329276Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.09252"},"observation_digest":"sha256:b3dc5bbf09787f8ee2b440b00be18cff1342a03401174f661c44b2062408cb1b","observation_id":"df52b1ab-c41e-4b5a-a125-ec598d19c1a2","resolution":{"observed_at":"2026-05-12T02:46:19.060948Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.09252","last_updated":"2026-05-20T19:57:19Z","snapshot_observed_at":"2026-08-02T16:03:49.427921Z","submitted_at":"2026-05-10T01:37:40Z","title":"LLM Agents Already Know When to Call Tools -- Even Without Reasoning","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-05-22T10:55:14.037226Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.09252"},"observation_digest":"sha256:16dd445414fac9924644892b8eefeb5f66dec24318884c76ed9e1fe761843293","observation_id":"f2f9c64f-d445-40d5-a80f-22e192a3ff2b","resolution":{"observed_at":"2026-05-22T10:56:25.965009Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.10365","last_updated":"2026-05-11T11:09:04Z","snapshot_observed_at":"2026-07-06T23:22:18.704329Z","submitted_at":"2026-05-11T11:09:04Z","title":"Agent-ValueBench: A Comprehensive Benchmark for Evaluating Agent Values","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-05-12T04:22:06.172150Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.10365"},"observation_digest":"sha256:5efd952d4b0e9c7ee45bcadbad0400791595300a80cf7ace357c3c7ea6d57637","observation_id":"f3f32258-9907-456a-953e-b8350f686e5c","resolution":{"observed_at":"2026-05-12T04:31:21.357721Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.10516","last_updated":"2026-05-11T13:06:24Z","snapshot_observed_at":"2026-07-06T23:22:28.318343Z","submitted_at":"2026-05-11T13:06:24Z","title":"Consistency as a Testable Property: Statistical Methods to Evaluate AI Agent Reliability","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-12T04:41:15.286881Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.10516"},"observation_digest":"sha256:fc10a57a7895fb53780d8824521ecdfa1fed0f6f64f545521dd930478d282293","observation_id":"e730d03d-012e-46db-ba61-4682e42e3a8a","resolution":{"observed_at":"2026-05-12T04:41:21.856981Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.10912","last_updated":"2026-05-11T17:49:43Z","snapshot_observed_at":"2026-08-03T00:15:01.823825Z","submitted_at":"2026-05-11T17:49:43Z","title":"WildClawBench: A Benchmark for Real-World, Long-Horizon Agent Evaluation","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-12T03:40:00.725327Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.10912"},"observation_digest":"sha256:dff4b6c380560963e5732f4df2f5df87b915ba8ed1dbbad8226523d1e640311e","observation_id":"bf81f40f-85c9-4ef1-b7a0-a9afd75ccfb4","resolution":{"observed_at":"2026-05-12T07:11:23.964559Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.10913","last_updated":"2026-06-24T17:55:07Z","snapshot_observed_at":"2026-08-02T06:55:55.998448Z","submitted_at":"2026-05-11T17:50:51Z","title":"Shepherd: Enabling Programmable Meta-Agents via Reversible Agentic Execution Traces","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-12T03:29:33.497561Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.10913"},"observation_digest":"sha256:e1ee6acab2d9486100c00503d715f683ae23c8fbae54cde2a283cfd48ce5218e","observation_id":"6ff4237c-23a0-432b-b0dc-b69a0e5b4df1","resolution":{"observed_at":"2026-05-12T07:16:31.018863Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.10913","last_updated":"2026-06-24T17:55:07Z","snapshot_observed_at":"2026-08-02T06:55:55.998448Z","submitted_at":"2026-05-11T17:50:51Z","title":"Shepherd: Enabling Programmable Meta-Agents via Reversible Agentic Execution Traces","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-30T22:24:26.528532Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.10913"},"observation_digest":"sha256:a5b9248c8ae56fe353c56d0215e76d8e60a30fbc52933cae7ac1e79449c1a82d","observation_id":"fca59e70-5b37-42a0-81e0-55231c125552","resolution":{"observed_at":"2026-06-30T22:25:06.838406Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.10966","last_updated":"2026-05-08T10:57:19Z","snapshot_observed_at":"2026-07-06T23:22:52.369277Z","submitted_at":"2026-05-08T10:57:19Z","title":"MMTB: Evaluating Terminal Agents on Multimedia-File Tasks","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-13T01:03:22.390574Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.10966"},"observation_digest":"sha256:3c16d4dfdc7c80de4a284c6d98fb51715b1cd8a6d66a5c6c31f52c06748d5403","observation_id":"51e3bf8a-4d6a-40bd-827e-3b03fabf1c24","resolution":{"observed_at":"2026-05-13T01:07:00.481516Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.11269","last_updated":"2026-05-11T21:47:22Z","snapshot_observed_at":"2026-07-06T23:23:06.755816Z","submitted_at":"2026-05-11T21:47:22Z","title":"gwBenchmarks: Stress-Testing LLM Agents on High-Precision Gravitational Wave Astronomy","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-13T02:14:48.091245Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.11269"},"observation_digest":"sha256:836ea8d8c91b566849fc28707cce27697f4181f29ee310e4ae93960cf10b2156","observation_id":"a5c385cf-7725-43df-88d9-56073daa5181","resolution":{"observed_at":"2026-05-13T02:17:06.452705Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-02T02:31:32.394072Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:7ce1fc543c94810d177914f5a0923ad4a1e64116ab68895b72d4bd7bb62d523d","observation_id":"c32812f1-7fb8-42ee-940e-4ce580779882","resolution":{"observed_at":"2026-05-14T20:32:56.949393Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.12925","last_updated":"2026-06-02T15:49:40Z","snapshot_observed_at":"2026-07-06T23:24:32.412966Z","submitted_at":"2026-05-13T03:00:57Z","title":"AgentLens: Revealing The Lucky Pass Problem in SWE-Agent Evaluation","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-05-14T18:51:06.379266Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.12925"},"observation_digest":"sha256:3b1068456dc5ff232e2d76cfe433791d3a75b1af3686d5885f903e5e11d47e8e","observation_id":"503b79ae-787f-49fa-83da-4f68cb2e5327","resolution":{"observed_at":"2026-05-14T18:52:35.464751Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.13880","last_updated":"2026-05-11T04:34:43Z","snapshot_observed_at":"2026-07-06T23:25:25.361350Z","submitted_at":"2026-05-11T04:34:43Z","title":"PREPING: Building Agent Memory without Tasks","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-15T06:14:14.385586Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.13880"},"observation_digest":"sha256:ded3cd4162c1901c6f373103e992f66b4966158b4428cdf09eec7c7f9e5c7f41","observation_id":"15c81c16-6fee-4891-adf5-99f8d6bc80d6","resolution":{"observed_at":"2026-05-15T06:15:06.120258Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.13950","last_updated":"2026-05-13T18:00:00Z","snapshot_observed_at":"2026-08-01T16:09:17.962725Z","submitted_at":"2026-05-13T18:00:00Z","title":"Collider-Bench: Benchmarking AI Agents with Particle Physics Analysis Reproduction","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-05-15T06:04:03.605898Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.13950"},"observation_digest":"sha256:624f72fb08210b452b364bd129ac23183a9f1d9f1af7d9279c8742ff50a7bf27","observation_id":"309e8e6a-3dae-44f3-8ced-fbfd0fda3b69","resolution":{"observed_at":"2026-05-15T06:05:06.740995Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.14084","last_updated":"2026-06-09T19:01:29Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T20:09:35Z","title":"CRANE: Constrained Reasoning Injection for Code Agents via Nullspace Editing","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-15T04:45:39.641535Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.14084"},"observation_digest":"sha256:9659140d1c44b231524d8585660d46c0af8f475335ea8991bf7c6835893388a0","observation_id":"4fc34411-3e50-43d2-ba18-160b02c5230f","resolution":{"observed_at":"2026-05-15T04:49:44.564961Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.14084","last_updated":"2026-06-09T19:01:29Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T20:09:35Z","title":"CRANE: Constrained Reasoning Injection for Code Agents via Nullspace Editing","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-30T21:12:37.353935Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.14084"},"observation_digest":"sha256:0a3263ea90bdb0ece4138f4e4f527b7e9ab103d2359d35b32876a175915880fc","observation_id":"aed8048e-41c0-4af3-892a-79f4cf2aa89d","resolution":{"observed_at":"2026-06-30T21:15:04.197628Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-15T05:08:15.750558Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:945b386d25a65e1438b9d2a050853cf03e87e1e9fbbb239dad0b7cd87d1144cb","observation_id":"78cfac98-5e6c-4228-a650-34cfaf220276","resolution":{"observed_at":"2026-05-15T05:09:45.015967Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:5b0c7300dd513dbe57caff05fb2d71ced1cdb82cdd1c51a6f9aec0c380a62036","observation_id":"6ea8b4e0-dd3e-471f-9a61-85c3ec0be6e3","resolution":{"observed_at":"2026-05-20T20:23:43.298529Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.14859","last_updated":"2026-05-15T03:53:20Z","snapshot_observed_at":"2026-07-06T23:26:13.644646Z","submitted_at":"2026-05-14T14:05:58Z","title":"Do Coding Agents Understand Least-Privilege Authorization?","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-19T16:34:14.379419Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.14859"},"observation_digest":"sha256:3b588091a80ad6b6138ce6db0e4270c90b2fc6a889f91f7eb2c2dbc70a6c782c","observation_id":"e219b491-3529-4255-91c5-73f8bd914e35","resolution":{"observed_at":"2026-05-19T16:37:40.123294Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.15040","last_updated":"2026-07-30T16:17:23Z","snapshot_observed_at":"2026-08-02T23:57:46.531676Z","submitted_at":"2026-05-14T16:35:12Z","title":"Orchard: An Open-Source Agentic Modeling Framework","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-07-11T11:50:26.030339Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.15040"},"observation_digest":"sha256:94d4b3e42d6fb771311e740127fd84dc76e77443664002173e8091ff67925a97","observation_id":"2f93dc2e-a6e9-43c4-845e-e4d0be8ab33a","resolution":{"observed_at":"2026-05-22T09:51:21.271005Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-02T14:05:15.131258Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2605.15040","last_updated":"2026-07-30T16:17:23Z","snapshot_observed_at":"2026-08-02T23:57:46.531676Z","submitted_at":"2026-05-14T16:35:12Z","title":"Orchard: An Open-Source Agentic Modeling Framework","version":3},"reference_index":2026,"source":"pdf_text","source_observed_at":"2026-08-02T14:05:15.131258Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.15040"},"observation_digest":"sha256:5d0bc5d4c1bc3a4e1854d01dba3d7bc3b6cc1cc264f2ef9f4d6a3b7e96f81da2","observation_id":"88082ec2-1c15-4900-a813-9175ecfadaf6","resolution":{"observed_at":"2026-08-02T14:05:15.131258Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.16508","last_updated":"2026-05-15T18:05:21Z","snapshot_observed_at":"2026-08-03T01:50:34.102829Z","submitted_at":"2026-05-15T18:05:21Z","title":"The Scaling Laws of Skills in LLM Agent Systems","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-20T18:10:08.737710Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.16508"},"observation_digest":"sha256:a4ed89e9b5380ecdaad2e3edf2caa40009dd458ac73e66c9f3b663ed9dcd6f20","observation_id":"27abc1ab-56ef-4fc6-9416-469e0595125f","resolution":{"observed_at":"2026-05-20T18:13:37.642484Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.16604","last_updated":"2026-05-15T20:10:24Z","snapshot_observed_at":"2026-07-06T23:27:43.762904Z","submitted_at":"2026-05-15T20:10:24Z","title":"R2V Agent: Teaching SLMs When to Ask for Help","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-20T20:26:29.804427Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.16604"},"observation_digest":"sha256:958b4f72e24244ae18513f2f1417b159f50d5ac4897a548fac70f4fa11786384","observation_id":"a69b9c14-c1b8-4e96-acff-34dcef2fff66","resolution":{"observed_at":"2026-05-20T20:28:59.778484Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.16679","last_updated":"2026-05-19T05:51:20Z","snapshot_observed_at":"2026-08-02T09:51:22.510077Z","submitted_at":"2026-05-15T22:34:31Z","title":"CHI-Bench: Can AI Agents Automate End-to-End, Long-Horizon, Policy-Rich Healthcare Workflows?","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-20T17:45:02.896703Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.16679"},"observation_digest":"sha256:6c37667a26be8cfeef9f309c0c904f1cd01154a426894e0f1445400abfabf5ac","observation_id":"75a26867-59a5-44f7-9907-2686bbfb54e2","resolution":{"observed_at":"2026-05-20T17:48:48.854593Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.17079","last_updated":"2026-05-16T16:55:31Z","snapshot_observed_at":"2026-07-06T23:28:06.971959Z","submitted_at":"2026-05-16T16:55:31Z","title":"Can LLMs Think Like Consumers? Benchmarking Crowd-Level Reaction Reconstruction with ConsumerSimBench","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-20T15:31:25.079191Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.17079"},"observation_digest":"sha256:a4eef380eb12303629016fc98e1e8a2fd373a33e66625915b83bbe763b1c727f","observation_id":"715c29b0-5284-465f-88ec-2763e48ffdd8","resolution":{"observed_at":"2026-05-20T15:33:25.519246Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.17169","last_updated":"2026-05-16T21:56:33Z","snapshot_observed_at":"2026-07-06T23:28:12.114488Z","submitted_at":"2026-05-16T21:56:33Z","title":"Responsible Agentic AI Requires Explicit Provenance","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-20T14:09:17.469239Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.17169"},"observation_digest":"sha256:589db912c17406d425a13ce21d93bdf07dcc4d2c9af96e4a10a9c719925e3d46","observation_id":"45f591ad-0cee-4f48-871a-0c59c6c30f26","resolution":{"observed_at":"2026-05-20T14:13:21.291811Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.17637","last_updated":"2026-05-21T20:02:32Z","snapshot_observed_at":"2026-08-03T12:30:05.317561Z","submitted_at":"2026-05-17T20:07:12Z","title":"WebGameBench: Requirement-to-Application Evaluation for Coding Agents via Browser-Native Games","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-20T12:24:06.062957Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.17637"},"observation_digest":"sha256:290106d86ba5f57bed635a9762b2c772714078d6aad9133f871e11d22065dc72","observation_id":"532443f7-28e3-4935-ac72-dc09c9cf633a","resolution":{"observed_at":"2026-05-20T12:28:17.276478Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.17637","last_updated":"2026-05-21T20:02:32Z","snapshot_observed_at":"2026-08-03T12:30:05.317561Z","submitted_at":"2026-05-17T20:07:12Z","title":"WebGameBench: Requirement-to-Application Evaluation for Coding Agents via Browser-Native Games","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-25T05:45:04.573722Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.17637"},"observation_digest":"sha256:b172b785f590a468cc1b44f1fcb3eb93a3079e7d7a6524c4f6e924ccdc48e370","observation_id":"93d737d9-6e7b-42b8-aaeb-c649ceb56585","resolution":{"observed_at":"2026-05-25T05:45:23.215403Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.18401","last_updated":"2026-05-18T13:44:19Z","snapshot_observed_at":"2026-07-06T23:29:20.279470Z","submitted_at":"2026-05-18T13:44:19Z","title":"SkillsVote: Lifecycle Governance of Agent Skills from Collection, Recommendation to Evolution","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-20T11:40:45.397038Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.18401"},"observation_digest":"sha256:e6c4b45c97a6e553e32e076ea87089f74769badd2c2abe2c5ad8e86be29eb067","observation_id":"ce1150a4-67d4-4813-9d06-6f50130693db","resolution":{"observed_at":"2026-05-20T11:43:15.124799Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.19341","last_updated":"2026-05-19T04:29:03Z","snapshot_observed_at":"2026-08-02T09:32:52.742127Z","submitted_at":"2026-05-19T04:29:03Z","title":"HalluWorld: A Controlled Benchmark for Hallucination via Reference World Models","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-20T05:52:53.880521Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.19341"},"observation_digest":"sha256:a4ba17a689da853260edcd9bf07b8cc341d837314bc0f37a670c28f916c32686","observation_id":"e97fe467-d648-43b5-b617-064f30d817af","resolution":{"observed_at":"2026-05-20T05:53:04.302217Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.20520","last_updated":"2026-05-19T21:42:32Z","snapshot_observed_at":"2026-08-03T02:30:24.579386Z","submitted_at":"2026-05-19T21:42:32Z","title":"Open-World Evaluations for Measuring Frontier AI Capabilities","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-21T06:38:51.427985Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.20520"},"observation_digest":"sha256:d8fa9ff3ceae6c8d283c12a1734541f3272eb33ad77ff3afe65e6a260cea6810","observation_id":"3cc47777-f3ce-4c2a-acd6-9db4470922e5","resolution":{"observed_at":"2026-05-21T06:39:43.684865Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.20744","last_updated":"2026-05-20T05:46:52Z","snapshot_observed_at":"2026-08-01T19:34:25.326316Z","submitted_at":"2026-05-20T05:46:52Z","title":"Hack-Verifiable Environments: Towards Evaluating Reward Hacking at Scale","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-21T06:56:27.532299Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.20744"},"observation_digest":"sha256:9b434b3a9d7f61267cc6d668855024608827f6ae645ed68c34b1c0a84a26409a","observation_id":"c3805acd-f5f9-4626-8013-2009cb555e79","resolution":{"observed_at":"2026-05-21T06:59:45.555982Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.20876","last_updated":"2026-05-20T08:14:51Z","snapshot_observed_at":"2026-07-06T23:31:23.407780Z","submitted_at":"2026-05-20T08:14:51Z","title":"Terminal-World: Scaling Terminal-Agent Environments via Agent Skills","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-21T04:54:17.662339Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.20876"},"observation_digest":"sha256:3ab62fc890802da148ef2f40820c98676867abf0f3cb47bb3029a8e21e69698c","observation_id":"9603e92f-7fe6-420b-985b-9c7047c38e74","resolution":{"observed_at":"2026-05-21T04:54:35.887170Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.21347","last_updated":"2026-06-04T23:27:42Z","snapshot_observed_at":"2026-07-06T23:31:46.877766Z","submitted_at":"2026-05-20T16:13:53Z","title":"Insights Generator: Systematic Corpus-Level Trace Diagnostics for LLM Agents","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-21T04:20:43.780849Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.21347"},"observation_digest":"sha256:3ab52b2ad8fcb3e24b23cf3850912a2e52fbc157c6a9147cceeb260fb990e14f","observation_id":"0548592f-515c-49c2-a65e-efb64215ce0e","resolution":{"observed_at":"2026-05-21T04:23:57.572966Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.21347","last_updated":"2026-06-04T23:27:42Z","snapshot_observed_at":"2026-07-06T23:31:46.877766Z","submitted_at":"2026-05-20T16:13:53Z","title":"Insights Generator: Systematic Corpus-Level Trace Diagnostics for LLM Agents","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-22T09:46:26.124683Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.21347"},"observation_digest":"sha256:c618480492287bf0f0c0206395ffd76b0961f7d7332c4c5bf5e56860bc027fe3","observation_id":"4c573288-681b-4aeb-a22e-3feb1f2a2278","resolution":{"observed_at":"2026-05-22T09:51:21.850352Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.21347","last_updated":"2026-06-04T23:27:42Z","snapshot_observed_at":"2026-07-06T23:31:46.877766Z","submitted_at":"2026-05-20T16:13:53Z","title":"Insights Generator: Systematic Corpus-Level Trace Diagnostics for LLM Agents","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-30T17:31:25.538977Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.21347"},"observation_digest":"sha256:f07d5eac0f70cd346eb1522ab597bac02e988365b94310c4073045abfdbda841","observation_id":"68cb9568-1160-4b29-bea1-9285dd28addd","resolution":{"observed_at":"2026-06-30T17:34:57.670499Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.22643","last_updated":"2026-05-22T14:53:30Z","snapshot_observed_at":"2026-07-06T23:32:59.663926Z","submitted_at":"2026-05-21T15:50:18Z","title":"Boiling the Frog: A Multi-Turn Benchmark for Agentic Safety","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-22T05:50:28.114140Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.22643"},"observation_digest":"sha256:3a4fd2314c7f53c2b7f5a101a8ab4018b61c8dbbca4a11fba90088c154360561","observation_id":"5ee85988-18d5-4098-a1f8-a56ac7e999f1","resolution":{"observed_at":"2026-05-22T05:51:07.764040Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.22643","last_updated":"2026-05-22T14:53:30Z","snapshot_observed_at":"2026-07-06T23:32:59.663926Z","submitted_at":"2026-05-21T15:50:18Z","title":"Boiling the Frog: A Multi-Turn Benchmark for Agentic Safety","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-25T06:05:27.736494Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.22643"},"observation_digest":"sha256:4ea777ceed201c6dfda168910e9dae2b1999766bf1ce80a079868ed0fc3b6690","observation_id":"de70236e-354d-44c0-a4b1-0b1eeace607b","resolution":{"observed_at":"2026-05-25T06:06:42.930484Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.23950","last_updated":"2026-05-07T15:24:59Z","snapshot_observed_at":"2026-08-03T13:44:55.519841Z","submitted_at":"2026-05-07T15:24:59Z","title":"Stop Comparing LLM Agents Without Disclosing the Harness","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-30T23:15:37.160073Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.23950"},"observation_digest":"sha256:b210883213f29528914269d0e495e17fdd4833ba121b3865ff0270b4830e68c5","observation_id":"12e2610c-841b-4ae3-a2c5-0ffe5acb0b16","resolution":{"observed_at":"2026-07-01T13:25:45.853016Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.24110","last_updated":"2026-05-22T18:17:28Z","snapshot_observed_at":"2026-08-04T22:48:29.574242Z","submitted_at":"2026-05-22T18:17:28Z","title":"EvoCode-Bench: Evaluating Coding Agents in Multi-Turn Iterative Interactions","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-11T11:50:26.030339Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.24110"},"observation_digest":"sha256:7e2455a1ff6ae45458a6e14f5a41774eb39b0f6a05501a9c46c3272823daeec3","observation_id":"c6e275ad-ec3e-4325-90da-d6c1b7f0fded","resolution":{"observed_at":"2026-06-30T16:24:53.896074Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.24117","last_updated":"2026-05-22T18:23:31Z","snapshot_observed_at":"2026-07-06T23:34:13.859813Z","submitted_at":"2026-05-22T18:23:31Z","title":"SkillEvolBench: Benchmarking the Evolution from Episodic Experience to Procedural Skills","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-30T16:18:59.083880Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.24117"},"observation_digest":"sha256:bc4576b53a177287f9ee9a94ef8edc54e2127fee05f407d54aec8a8943f91100","observation_id":"841dbef7-3a3c-43cb-ab9a-74663fcee331","resolution":{"observed_at":"2026-06-30T16:24:55.867127Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.24539","last_updated":"2026-05-23T12:10:52Z","snapshot_observed_at":"2026-07-06T23:34:38.098785Z","submitted_at":"2026-05-23T12:10:52Z","title":"DemoEvolve: Overcoming Sparse Feedback in Agentic Harness Evolution with Demonstrations","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-30T13:12:46.927103Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.24539"},"observation_digest":"sha256:3651898358c9ed7b2f2372a8c2a4bfc6b7ef2cc40d04a878684a520e5090c9d1","observation_id":"deefa0d1-ddb4-4fb0-8e1c-bc03ee987e64","resolution":{"observed_at":"2026-06-30T13:14:40.626642Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.25411","last_updated":"2026-05-25T04:23:29Z","snapshot_observed_at":"2026-07-06T23:35:21.734804Z","submitted_at":"2026-05-25T04:23:29Z","title":"Heimdall: Formally Verified Automated Migration of Legacy eBPF Programs to Rust","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-06-29T22:02:16.597531Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.25411"},"observation_digest":"sha256:59c9faa6a2c594ea6d8378b17a8152cb8d0ef54bef43038b67cbac2544aef5a8","observation_id":"28ff39d4-3f31-448f-8496-ced88e58a6ae","resolution":{"observed_at":"2026-06-29T22:04:00.140390Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.26112","last_updated":"2026-05-25T17:59:36Z","snapshot_observed_at":"2026-07-06T23:36:01.564747Z","submitted_at":"2026-05-25T17:59:36Z","title":"From Model Scaling to System Scaling: Scaling the Harness in Agentic AI","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-29T21:29:57.326718Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.26112"},"observation_digest":"sha256:318a775db8840a81b2b21049c690c12b7a9e8932fc4d7258431f1f9ba3815cdf","observation_id":"2a338a80-4f3c-4d43-95ad-779cfd6205db","resolution":{"observed_at":"2026-06-29T21:33:59.178143Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.26297","last_updated":"2026-05-25T19:45:21Z","snapshot_observed_at":"2026-08-03T04:38:08.316378Z","submitted_at":"2026-05-25T19:45:21Z","title":"Agentic AI Workload Characteristics","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-29T20:11:50.722787Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.26297"},"observation_digest":"sha256:9f266b43e94a620cab0a829a27f7a21bae6e01a7dbf9e739fc893ba4b3049f2a","observation_id":"fe3a6ea6-693b-4610-a129-dfce44b43809","resolution":{"observed_at":"2026-06-29T20:13:58.843019Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.27492","last_updated":"2026-05-26T16:28:10Z","snapshot_observed_at":"2026-07-31T21:45:32.197960Z","submitted_at":"2026-05-26T16:28:10Z","title":"Benchmarks are Not Enough: RAMP for Runtime Assessing of Agentic Models in Production Systems","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-29T15:32:21.737028Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.27492"},"observation_digest":"sha256:66e0d2a3fbfa9172dc9d0fb4ad45cb040e77010686d3e510d2e07cc9f1e70459","observation_id":"e274befe-b446-46bb-b4a5-9234ca378836","resolution":{"observed_at":"2026-06-29T15:33:32.776969Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.29559","last_updated":"2026-05-28T08:11:57Z","snapshot_observed_at":"2026-07-06T23:38:56.014434Z","submitted_at":"2026-05-28T08:11:57Z","title":"LiteCoder-Terminal: Scaling Long-Horizon Terminal Environments for Learning Language Agents","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-29T07:27:45.349782Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.29559"},"observation_digest":"sha256:5c0d6b4f82e69aa7ecabf967a1d3b5cd220618b77c834b8df1a64fd9e13ca827","observation_id":"bbb67e4c-f862-46d1-a376-9406eedf60ca","resolution":{"observed_at":"2026-06-29T07:33:13.901903Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.31308","last_updated":"2026-05-29T13:40:31Z","snapshot_observed_at":"2026-08-02T19:36:00.281512Z","submitted_at":"2026-05-29T13:40:31Z","title":"TraceGraph: Shared Decision Landscapes for Diagnosing and Improving Agent Trajectories","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-28T22:22:54.473952Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.31308"},"observation_digest":"sha256:6241c729f01470036e30868bcb227cf417f9f1e397ec5af0e36232e4a061333c","observation_id":"54a1d8d7-304f-474f-b75a-e6fb64032235","resolution":{"observed_at":"2026-07-01T19:36:08.589449Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.31593","last_updated":"2026-05-29T17:57:00Z","snapshot_observed_at":"2026-07-06T23:40:46.825645Z","submitted_at":"2026-05-29T17:57:00Z","title":"Stateful Online Monitoring Catches Distributed Agent Attacks","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-06-28T21:54:44.072929Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.31593"},"observation_digest":"sha256:86a6cf48593b79f9306390243abfd23fd52f5e2045b9d0508550dbd87e45bab4","observation_id":"a0b672bc-4ad4-429b-8d7d-b5e3125865f7","resolution":{"observed_at":"2026-07-01T19:56:11.080961Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2606.00341","last_updated":"2026-05-29T20:29:35Z","snapshot_observed_at":"2026-07-06T23:41:05.637982Z","submitted_at":"2026-05-29T20:29:35Z","title":"ROGUE: Misaligned Agent Behavior Arising from Ordinary Computer Use","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-28T23:03:36.851403Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2606.00341"},"observation_digest":"sha256:9fd6b48765ba30f5a4d809cb6140f65cd21e5b830324285573edd340d3e66674","observation_id":"c1ef4e1d-6e4b-4cb0-b36a-03adb6528df3","resolution":{"observed_at":"2026-07-01T19:16:00.188334Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2606.00530","last_updated":"2026-08-02T19:01:22Z","snapshot_observed_at":"2026-08-05T19:49:37.866954Z","submitted_at":"2026-05-30T04:49:10Z","title":"Sakura: An Approach for Generating Complex Tests from Natural Language Test Descriptions","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-06-28T18:40:49.564295Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2606.00530"},"observation_digest":"sha256:3496589266b272a975cf4106fc469ef6c352f3c4929ee16a4d807059239b7294","observation_id":"1404f36a-9048-437a-aee0-5f5b65eb72eb","resolution":{"observed_at":"2026-06-28T18:42:29.432773Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2601.11868/citation-record","integrity":"/paper/2601.11868/integrity","json":"/paper/2601.11868/citation-record.json","paper":"/paper/2601.11868"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2504.11442","last_updated":"2025-05-24T04:10:59Z","snapshot_observed_at":"2026-07-06T21:09:55.394184Z","submitted_at":"2025-04-15T17:55:20Z","title":"TextArena","version":2},"cited_work":{"arxiv_id":"2504.11442","doi":"10.48550/arxiv.2504.11442","metadata_source":"arxiv_reference","pith_arxiv_id":"2504.11442","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Textarena","venue":"ArXiv.org","work_id":"5e2b14af-8a88-4c5c-8bfd-831d9590aa34","year":2025},"citing_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-11T03:37:07.841385Z"},"links":{"cited_paper":"/paper/2504.11442","citing_paper":"/paper/2601.11868"},"observation_digest":"sha256:27083761712365659c007434f858d546d0ccec5ebf1598f6fc8d7d6c737cd3ee","observation_id":"ba6d4cc9-f1fd-4bd6-8dcc-cb5f6381aad1","resolution":{"observed_at":"2026-05-11T03:37:08.016661Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2024.acl-long.50","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VisualWebArena: Evaluating multimodal agents on realistic visual web tasks","venue":null,"work_id":"7a18db33-5a35-436c-ba83-41b4602aea68","year":2024},"citing_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-11T03:37:07.841385Z"},"links":{"citing_paper":"/paper/2601.11868"},"observation_digest":"sha256:ca20d5a80b9b915794d4ea15ed5f3420b73a2a1d496d6b96ee9a65c3365cac1c","observation_id":"cd9c0420-6168-4263-b307-9dde5ac8dbb6","resolution":{"observed_at":"2026-05-11T03:37:07.976590Z","resolver_source":"doi","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-07-15T19:20:32.353486+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-15T19:20:32.353486+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.10925","last_updated":"2025-08-08T19:24:38Z","snapshot_observed_at":"2026-08-01T16:27:35.664983Z","submitted_at":"2025-08-08T19:24:38Z","title":"gpt-oss-120b & gpt-oss-20b Model Card","version":1},"cited_work":{"arxiv_id":"2508.10925","doi":"10.3115/1073083.1073135","metadata_source":"pith","pith_arxiv_id":"2508.10925","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"gpt-oss-120b & gpt-oss-20b Model Card","venue":"cs.CL","work_id":"178c1f7e-4f19-4392-a45d-45a6dfa88ead","year":2025},"citing_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-11T03:37:07.841385Z"},"links":{"cited_paper":"/paper/2508.10925","citing_paper":"/paper/2601.11868"},"observation_digest":"sha256:c77c8b509aee7d75c9b211d835be6d41cc6d5f0ff5f874edc60d98a78978aacf","observation_id":"b485e30e-fadf-47a8-bb07-efe7236bdf3b","resolution":{"observed_at":"2026-05-11T03:37:08.005699Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-07-11T19:19:47.962081+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T19:19:47.962081+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2507.15887","doi":"10.48550/arxiv.2507.15887","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"K., Krupke, D., Kidger, P., Sajed, T., Stellato, B., Park, J., et al","venue":"ArXiv.org","work_id":"2c25f9d0-1696-48ff-a3b2-38c2e0d862c1","year":2025},"citing_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-11T03:37:07.841385Z"},"links":{"citing_paper":"/paper/2601.11868"},"observation_digest":"sha256:7dedaa7962e909bcbd9f7e093e139d80e0c4c987fa331fc17ce258497df8cf66","observation_id":"33ee05cc-1213-40ed-b33c-c6a79eff3c2c","resolution":{"observed_at":"2026-05-11T03:37:07.967629Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-07-06T21:51:52.256162Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":"2507.02825","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-07-03T22:08:59.930877Z","title":"Establishing best practices for building rigorous agentic benchmarks","venue":null,"work_id":"981c0760-99e5-4175-bca8-6be3ed159f5b","year":2025},"citing_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-11T03:37:07.841385Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2601.11868"},"observation_digest":"sha256:03533c20dbd79eb633b52704382c63036bd5b3ef9966f1e5523f0523587d62e5","observation_id":"66ecafb5-6e96-4e31-a056-9461b2744708","resolution":{"observed_at":"2026-05-11T03:37:07.992049Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"18 APPENDIXTABLE OFCONTENTS A Detailed Results 21 A.1 Comprehensive Results","venue":null,"work_id":"5dfdb7c1-ac9b-4585-82b5-c6e9cf4567dd","year":null},"citing_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-11T03:37:07.841385Z"},"links":{"citing_paper":"/paper/2601.11868"},"observation_digest":"sha256:9c7060e1f28344ad25e01bbaf2908136731d2539b5ae754b8c94fbec31af6c42","observation_id":"95239b1d-5d3b-44fe-8c52-63b01aefd6d2","resolution":{"observed_at":"2026-05-11T03:37:08.114457Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"b76c28e2-82cf-403d-aa96-7612846c93b5","year":null},"citing_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-11T03:37:07.841385Z"},"links":{"citing_paper":"/paper/2601.11868"},"observation_digest":"sha256:c3d01d2b275b5d9bc31bb039ece6702b62b164b70f129ea89906f728135a7d11","observation_id":"b7a596a1-1c32-4bbe-8ce8-93c5f9ed1886","resolution":{"observed_at":"2026-05-11T03:37:08.134349Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"9abb0f52-ddd8-4089-bf23-9c856a226e24","year":null},"citing_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-11T03:37:07.841385Z"},"links":{"citing_paper":"/paper/2601.11868"},"observation_digest":"sha256:bd5a77c983e40b77284821a61fbfddce826f43355c5863fc2e7324e568fd812f","observation_id":"e3bb58bb-4a28-4859-8580-3b1f89c34848","resolution":{"observed_at":"2026-05-11T03:37:08.142535Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"f9875b64-39c4-47da-99dc-1159599db064","year":null},"citing_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-11T03:37:07.841385Z"},"links":{"citing_paper":"/paper/2601.11868"},"observation_digest":"sha256:732993c040d84c32101ca6386742376b3a011401774f5ffcfbe7dcf109ba34d2","observation_id":"5f12f90b-f0f7-4a08-a6f0-9400bed77132","resolution":{"observed_at":"2026-05-11T03:37:08.146944Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"outcome\"","venue":null,"work_id":"b143a60c-0c78-4790-9525-81f4ec8fc96b","year":2025},"citing_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-11T03:37:07.841385Z"},"links":{"citing_paper":"/paper/2601.11868"},"observation_digest":"sha256:c3bfde0e15424e2c5f8d23f9bdcdc9e11a4a3c7f85024f9945df73de865c0332","observation_id":"1ea20868-973d-44d0-b83f-5744ccd0f5de","resolution":{"observed_at":"2026-05-11T03:37:08.028130Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"245a5d77-659c-45ab-8fdc-70662d173b00","year":null},"citing_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-11T03:37:07.841385Z"},"links":{"citing_paper":"/paper/2601.11868"},"observation_digest":"sha256:c5589aaef3c9c053798da6a7527f6222d61809d55aaa3be140e31ea16f725805","observation_id":"704a83f3-364a-48de-bd1a-fee2d68af6f1","resolution":{"observed_at":"2026-05-11T03:37:08.035125Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Independent inspection of the released tasks con- firmed that they are well-specified and largely free of ambiguity or underspecification","venue":null,"work_id":"b5ea545a-5a85-4a56-9266-014a4ad11ab8","year":null},"citing_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-11T03:37:07.841385Z"},"links":{"citing_paper":"/paper/2601.11868"},"observation_digest":"sha256:29992c3604b6e442b4dd0eaf883a51266a15b753795bc65090d1d7aa0ea031fa","observation_id":"18234bb5-8c76-4752-91b3-3aa121ccce72","resolution":{"observed_at":"2026-05-11T03:37:08.042998Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"held-out","venue":null,"work_id":"48210be2-e622-4862-81cb-67770d6309fc","year":2025},"citing_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-11T03:37:07.841385Z"},"links":{"citing_paper":"/paper/2601.11868"},"observation_digest":"sha256:8b33478ade0c08dfc919543841fa9ab3186c7e13fb4ccce92dd6a7001fd281f2","observation_id":"69736d52-131f-43c6-8a5e-72c1743023fe","resolution":{"observed_at":"2026-05-11T03:37:08.049443Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"7940947d-8498-4c16-8e19-1c42b6500ad8","year":null},"citing_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-11T03:37:07.841385Z"},"links":{"citing_paper":"/paper/2601.11868"},"observation_digest":"sha256:21cd491c0bef2c44f6cc8158d29753247edfb51673756f43da8c350c69a8176c","observation_id":"8420948f-997f-420e-8658-33e85060d913","resolution":{"observed_at":"2026-05-11T03:37:08.055164Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"We’re looking for indications that running the command has failed - carefully analyse the outputs to determine this","venue":null,"work_id":"4d551179-1d90-438e-8f12-d65744f19967","year":null},"citing_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-11T03:37:07.841385Z"},"links":{"citing_paper":"/paper/2601.11868"},"observation_digest":"sha256:84c26b756b01889c5af99ab33dfb7df95755b97b168c9a846a64c7bc1c9bb14d","observation_id":"79ae3ee4-b721-4884-a081-76231d0e48a3","resolution":{"observed_at":"2026-05-11T03:37:08.061385Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"8ef752c3-5290-4944-86f9-07a358d8e7f8","year":null},"citing_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-11T03:37:07.841385Z"},"links":{"citing_paper":"/paper/2601.11868"},"observation_digest":"sha256:4622ff455a587d742abac6e65ffe3918fa558b05a1db60334ac73ebcfe0640e2","observation_id":"b78106c3-ece6-4f01-8d47-5bb24889ec82","resolution":{"observed_at":"2026-05-11T03:37:08.068234Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"a2f92df6-28cb-4d33-9397-bb44efda832e","year":null},"citing_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-11T03:37:07.841385Z"},"links":{"citing_paper":"/paper/2601.11868"},"observation_digest":"sha256:2bdc6bd5a398f37e320a09d029770b9b9cfb8b7e7b9e6e07e8dedbf1f907ece4","observation_id":"df70fb1f-1c80-445d-b7c3-83b63bab7be6","resolution":{"observed_at":"2026-05-11T03:37:08.074051Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"You are about to hand off your work to another agent. Please summarize what you’ve done so far","venue":null,"work_id":"cec1483f-6c97-4e8e-ac7f-6eb6611d6561","year":null},"citing_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-11T03:37:07.841385Z"},"links":{"citing_paper":"/paper/2601.11868"},"observation_digest":"sha256:6a61ce3e0575c9db674770fe710a54b9a732bcecd309bc32d6245372a15c30d9","observation_id":"60003a15-e549-4f65-bdbe-2021236c8a53","resolution":{"observed_at":"2026-05-11T03:37:08.088349Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"1ee220eb-3c8b-42c0-816d-9431c2481097","year":null},"citing_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-11T03:37:07.841385Z"},"links":{"citing_paper":"/paper/2601.11868"},"observation_digest":"sha256:c003cde72febab40942f662c36bceb34029896064ee74b9cfd0a637c9c4693ae","observation_id":"c92b3ab6-065e-40ca-9339-c9cd25485b5a","resolution":{"observed_at":"2026-05-11T03:37:08.101181Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Results: X Y Z","venue":null,"work_id":"f508bd91-fe37-45bd-9535-27f7943cf9a9","year":1992},"citing_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-11T03:37:07.841385Z"},"links":{"citing_paper":"/paper/2601.11868"},"observation_digest":"sha256:d5d28d1e6af585ae4d533ae8e0d1776e10dff7152eeab7ba8e11eda3e4337016","observation_id":"75a8499d-5053-419f-8936-085fa367be89","resolution":{"observed_at":"2026-05-11T03:37:08.109352Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","latest_version":1,"primary_category":"cs.SE","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces"},"reference_resolution":{"displayed":20,"state_counts":{"malformed_identifier":2,"metadata_mismatch":4,"parse_uncertain":0,"unresolved":8,"verified_exact":1,"verified_fuzzy":5},"total_outbound_references":20},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 5 August 2026, this Paper Citation Record lists 20 of 20 outbound references and 100 inbound Pith citation observations for arXiv:2601.11868."}