{"as_of":"2026-08-06T21:35:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:f605c5ebc66cf1deb9c07cc748eb53c4f61115c6c499e7bfc28c44ab1c715abc","coverage":[{"denominator":24,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":24,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-12T13:48:53.691192Z","state":"measured"},{"denominator":124,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":124,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":153,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-03T20:17:17.509512Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-03T20:17:17.509512Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2511.20613","last_updated":"2025-11-25T18:40:22Z","snapshot_observed_at":"2026-08-06T16:18:03.724570Z","submitted_at":"2025-11-25T18:40:22Z","title":"Can Vibe Coding Beat Graduate CS Students? An LLM vs. Human Coding Tournament on Market-driven Strategic Planning","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-03T20:17:17.509512Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2511.20613"},"observation_digest":"sha256:9faee72f4879b0a75e49049a0b75e3915199cb781d1f4a0476c5891c9f70915a","observation_id":"13f16d5c-f481-4a6f-a3d1-39ca7c7ba0f7","resolution":{"observed_at":"2026-08-03T20:17:17.509512Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2512.18470","last_updated":"2026-04-04T09:52:04Z","snapshot_observed_at":"2026-07-30T13:30:43.108201Z","submitted_at":"2025-12-20T19:08:15Z","title":"SWE-EVO: Benchmarking Coding Agents in Long-Horizon Software Evolution Scenarios","version":5},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-05-16T20:24:40.939455Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2512.18470"},"observation_digest":"sha256:30cf41d198b97b34b5f63895d62e3de14797014ac29a711494cf5e8d5373bdf0","observation_id":"a7e9f8cb-82d3-4683-8574-95f7359de454","resolution":{"observed_at":"2026-05-16T20:28:24.562895Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2512.18552","last_updated":"2026-06-02T06:06:43Z","snapshot_observed_at":"2026-08-03T15:02:08.953642Z","submitted_at":"2025-12-21T00:49:40Z","title":"Toward Training Superintelligent Software Agents through Self-Play SWE-RL","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-21T16:07:48.570995Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2512.18552"},"observation_digest":"sha256:a8f4deed2784ae83dd1148e62bd5c3a58814915f60edb3a2d9372c3ff1ffab5f","observation_id":"f06ad605-bd3d-4005-a720-f5c4fbdb8310","resolution":{"observed_at":"2026-05-21T16:10:20.250611Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-03T15:02:10.639537Z","title":"arXiv: 2509.16941 [cs.SE]","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.18552","last_updated":"2026-06-02T06:06:43Z","snapshot_observed_at":"2026-08-03T15:02:08.953642Z","submitted_at":"2025-12-21T00:49:40Z","title":"Toward Training Superintelligent Software Agents through Self-Play SWE-RL","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-03T15:02:10.639537Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2512.18552"},"observation_digest":"sha256:ae59e7a40b29a5dc894d8e6fd19b523c5b75e084b62ec020130d395dec400d76","observation_id":"3173a31c-9f42-4acb-a605-f8e95113e822","resolution":{"observed_at":"2026-08-03T15:02:10.639537Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2601.05106","last_updated":"2026-05-21T03:21:52Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-01-08T16:53:16Z","title":"Token-Level LLM Collaboration via FusionRoute","version":4},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-22T12:25:59.747665Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2601.05106"},"observation_digest":"sha256:601645dfe3a082535c33fa8880350a75851260541ceb53bf3a3ded476d36d12b","observation_id":"b595c520-873a-4502-89d0-04b38fa983e0","resolution":{"observed_at":"2026-05-22T12:26:31.643260Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2602.02276","last_updated":"2026-02-02T16:17:38Z","snapshot_observed_at":"2026-07-06T22:44:09.804048Z","submitted_at":"2026-02-02T16:17:38Z","title":"Kimi K2.5: Visual Agentic Intelligence","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-10T16:09:05.225767Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2602.02276"},"observation_digest":"sha256:83d697af62a80c711ca588ae7435b9dd78f972cfdf4f181c8694310ff894ee13","observation_id":"9ce49000-131b-4fba-af4c-00b792727133","resolution":{"observed_at":"2026-05-12T13:48:53.849224Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2602.11224","last_updated":"2026-04-28T16:08:25Z","snapshot_observed_at":"2026-07-06T22:45:29.609382Z","submitted_at":"2026-02-11T13:31:18Z","title":"Agent-Diff: Benchmarking LLM Agents on Enterprise API Tasks via Code Execution with State-Diff-Based Evaluation","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-16T03:04:17.755968Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2602.11224"},"observation_digest":"sha256:9371b5cb20430aca2915332aab425f6bd989d115f8ac561672732cdd6254b0d8","observation_id":"4d4d8e91-4238-43a2-bc70-f90979d13cbc","resolution":{"observed_at":"2026-05-16T03:07:11.979895Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-02T22:26:35.274592Z","title":"Swe-bench pro: Can ai agents solve long-horizon software engineering tasks? arXiv preprint arXiv:2509.16941, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.16902","last_updated":"2026-05-30T16:45:37Z","snapshot_observed_at":"2026-08-06T05:35:47.226397Z","submitted_at":"2026-02-18T21:33:59Z","title":"LLM-WikiRace Benchmark: How Far Can LLMs Plan over Real-World Knowledge Graphs?","version":4},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-02T22:26:35.274592Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2602.16902"},"observation_digest":"sha256:fcea686c11ac4b2d82ad09e117939f1411480bc8140e7b9170a05ecfa52028a0","observation_id":"efab4aec-579f-4fb9-8c2b-1a25fe507601","resolution":{"observed_at":"2026-08-02T22:26:35.274592Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2602.18571","last_updated":"2026-04-21T09:32:53Z","snapshot_observed_at":"2026-07-06T22:46:35.212918Z","submitted_at":"2026-02-20T19:24:16Z","title":"Debug2Fix: Can Interactive Debugging Help Coding Agents Fix More Bugs?","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-15T20:17:52.610119Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2602.18571"},"observation_digest":"sha256:b7a990367221839c8853e74cc5b25905e8da77367716501a1139e446b5eca026","observation_id":"72356ece-1253-4191-8d6d-7e65b1af3de5","resolution":{"observed_at":"2026-05-15T20:20:17.840575Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-02T19:43:15.515948Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.01327","last_updated":"2026-05-25T21:06:42Z","snapshot_observed_at":"2026-08-06T20:26:09.749875Z","submitted_at":"2026-03-01T23:52:30Z","title":"SWE-Adept: An LLM-Based Agentic Framework for Deep Codebase Analysis and Structured Issue Resolution","version":2},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-02T19:43:15.515948Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2603.01327"},"observation_digest":"sha256:dcceb415a4edd3d55090738b23721f16dd212fc3dba5266f388ee9a882eb2d4a","observation_id":"857d370e-323e-4cbd-9d14-fc454972b0e1","resolution":{"observed_at":"2026-08-02T19:43:15.515948Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-02T19:12:54.855092Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.03194","last_updated":"2026-05-26T12:00:46Z","snapshot_observed_at":"2026-08-03T23:49:54.364126Z","submitted_at":"2026-03-03T17:52:01Z","title":"BeyondSWE: Can Current Code Agent Survive Beyond Single-Repo Bug Fixing?","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-02T19:12:54.855092Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2603.03194"},"observation_digest":"sha256:41ba773554ab8f60acadbe86cd35e22d17a9b5b7ae6cc3147fefd0b175049f59","observation_id":"c877ba6d-c61a-4fee-9e42-e86dba31844d","resolution":{"observed_at":"2026-08-02T19:12:54.855092Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2603.04601","last_updated":"2026-05-13T23:00:10Z","snapshot_observed_at":"2026-07-06T22:47:55.281598Z","submitted_at":"2026-03-04T21:00:33Z","title":"Vibe Code Bench: Evaluating AI Models on End-to-End Web Application Development","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-15T15:59:29.910200Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2603.04601"},"observation_digest":"sha256:1b88fe386d2a75ffeccf4e715998d0549a6f951d9ae08ccf89fb082ee82bd2a1","observation_id":"11f9be90-3195-48cf-9e38-bc5bf5c83ee4","resolution":{"observed_at":"2026-05-15T16:00:09.494253Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-02T18:20:29.593108Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.13428","last_updated":"2026-07-21T11:31:57Z","snapshot_observed_at":"2026-08-03T16:57:27.649639Z","submitted_at":"2026-03-13T03:20:40Z","title":"SWE-Milestone: Evaluating AI Agents on Continuous Software Evolution","version":4},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-02T18:20:29.593108Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2603.13428"},"observation_digest":"sha256:3add7c223f67e54f0039d6f598180fd9e2d1d18173d7ffd7cb4db8be86f49fb6","observation_id":"ac9ac8a3-bcd2-4802-82da-6601d83b922a","resolution":{"observed_at":"2026-08-02T18:20:29.593108Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-07-13T20:49:09.477849Z","title":"Xiang Deng, Jeff Da, Edwin Pan, Yannis Yiming He, Charles Ide, Kanak Garg, Niklas Lauffer, Andrew Park, Nitin Pasari, Chetan Rane, et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.21489","last_updated":"2026-07-08T16:30:55Z","snapshot_observed_at":"2026-08-03T00:37:21.202769Z","submitted_at":"2026-03-23T02:26:35Z","title":"Effective Strategies for Asynchronous Software Engineering Agents","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-07-13T20:49:09.477849Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2603.21489"},"observation_digest":"sha256:a71ae92f56bda15b8f857827f401b5ff9dd43695b214cb44fe71f524784ceae1","observation_id":"962a08fc-1c02-42ef-9fb9-073405967f2e","resolution":{"observed_at":"2026-07-13T20:49:09.477849Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-07-13T20:08:17.066898Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.22744","last_updated":"2026-05-29T00:35:02Z","snapshot_observed_at":"2026-08-03T18:04:48.244219Z","submitted_at":"2026-03-24T03:16:32Z","title":"LH-Bench: Skill-Grounded Evaluation of Long-Horizon Agents on Subjective Enterprise Tasks","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-13T20:08:17.066898Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2603.22744"},"observation_digest":"sha256:91c53a6f6e39ba40e9d66521ceb83c15d2d6228708af30de20f7f378b1ceaec9","observation_id":"1ca76b21-4b78-47cd-b3b0-9599ad8d659b","resolution":{"observed_at":"2026-07-13T20:08:17.066898Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-02T17:05:04.619944Z","title":"hello\") plt.subplot(3, 3, 4) plt.imshow(data) + plt.ylabel(","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2604.00594","last_updated":"2026-07-20T11:38:10Z","snapshot_observed_at":"2026-08-06T19:53:16.363700Z","submitted_at":"2026-04-01T07:59:59Z","title":"Agent psychometrics: Task-level performance prediction in agentic coding benchmarks","version":2},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-02T17:05:04.619944Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2604.00594"},"observation_digest":"sha256:5966b6943abe0bd2cbaa0e5eef57829a4ce1aeb6043e39c8cfe03f9ab482751c","observation_id":"a319e9c7-0805-4589-ac66-9ef7b6cdb150","resolution":{"observed_at":"2026-08-02T17:05:04.619944Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2604.02547","last_updated":"2026-04-02T21:56:23Z","snapshot_observed_at":"2026-08-03T04:47:13.899796Z","submitted_at":"2026-04-02T21:56:23Z","title":"Beyond Resolution Rates: Behavioral Drivers of Coding Agent Success and Failure","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-13T20:30:40.124360Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2604.02547"},"observation_digest":"sha256:d7faf1be7472410c3ea4a14497b50050b3b53deaecf71d704c50ffbddb9a1b84","observation_id":"10f2fbd2-f229-4841-b32c-c2e956e3fbb9","resolution":{"observed_at":"2026-05-13T20:33:16.411149Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2604.02947","last_updated":"2026-04-03T10:29:31Z","snapshot_observed_at":"2026-07-29T16:09:43.498926Z","submitted_at":"2026-04-03T10:29:31Z","title":"AgentHazard: A Benchmark for Evaluating Harmful Behavior in Computer-Use Agents","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-13T19:42:53.778980Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2604.02947"},"observation_digest":"sha256:cb4deca66abcd0b56b6bfffbd254fd32d26b2a05a7a2cd242ae5c176993585b3","observation_id":"5fe1d313-cf97-414c-9b64-77cc0c915ceb","resolution":{"observed_at":"2026-05-13T19:43:11.189056Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2604.03515","last_updated":"2026-04-10T20:06:33Z","snapshot_observed_at":"2026-07-06T22:52:39.423471Z","submitted_at":"2026-04-03T23:30:02Z","title":"Inside the Scaffold: A Source-Code Taxonomy of Coding Agent Architectures","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-13T18:03:18.475429Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2604.03515"},"observation_digest":"sha256:8d552a052699fb82ac578285cb0586052d7400c00ecd8cbaf678c58f899e1df9","observation_id":"63be1b4f-e3d6-4aba-8384-0395a19e2e57","resolution":{"observed_at":"2026-05-13T18:08:15.842530Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2604.05955","last_updated":"2026-04-07T14:47:27Z","snapshot_observed_at":"2026-07-06T22:54:34.877944Z","submitted_at":"2026-04-07T14:47:27Z","title":"Does Pass Rate Tell the Whole Story? Evaluating Design Constraint Compliance in LLM-based Issue Resolution","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-10T18:45:06.682021Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2604.05955"},"observation_digest":"sha256:abe208f0c66cf99891b1e77da636ef10f5782ac79372a8627a8958a88dc6f458","observation_id":"58ff9215-614e-4895-9c28-56911a4cc6cf","resolution":{"observed_at":"2026-05-12T13:48:53.849224Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2604.06111","last_updated":"2026-04-10T03:07:44Z","snapshot_observed_at":"2026-08-02T21:04:18.191344Z","submitted_at":"2026-04-07T17:21:28Z","title":"AgentCE-Bench: Agent Configurable Evaluation with Scalable Horizons and Controllable Difficulty under Lightweight Environments","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T19:07:46.077831Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2604.06111"},"observation_digest":"sha256:200bd47a10ab10bcdec73ed9bef943207f5ef2480602855ef354181872eff8f0","observation_id":"03ad9450-e42e-466d-8228-29516e394f06","resolution":{"observed_at":"2026-05-12T13:48:53.849224Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2604.06861","last_updated":"2026-04-08T09:22:30Z","snapshot_observed_at":"2026-08-05T16:01:05.723535Z","submitted_at":"2026-04-08T09:22:30Z","title":"REAgent: Requirement-Driven LLM Agents for Software Issue Resolution","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-10T17:56:32.201591Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2604.06861"},"observation_digest":"sha256:ce127ac201da177d9cc7cf97242cba895b7f307fea8eef0b7679580f2a97d02c","observation_id":"b677eb3d-a0bf-4cce-a88f-80d918952e7e","resolution":{"observed_at":"2026-05-12T13:48:53.849224Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-07-13T00:15:09.034899Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2604.07789","last_updated":"2026-05-28T09:34:24Z","snapshot_observed_at":"2026-08-05T11:25:41.218088Z","submitted_at":"2026-04-09T04:37:24Z","title":"ORACLE-SWE: Quantifying the Contribution of Oracle Information Signals on SWE Agents","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-07-13T00:15:09.034899Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2604.07789"},"observation_digest":"sha256:7e8fa7725c51f2e530a5685cf7353fc7ae65354a4d47b6b9b1a400f6c27e7286","observation_id":"d6077b97-22f5-4326-a9c8-a3fd6521f465","resolution":{"observed_at":"2026-07-13T00:15:09.034899Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2604.09408","last_updated":"2026-05-04T23:00:33Z","snapshot_observed_at":"2026-07-06T22:58:17.167367Z","submitted_at":"2026-04-10T15:21:44Z","title":"HiL-Bench (Human-in-Loop Benchmark): Do Agents Know When to Ask for Help?","version":4},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-10T18:20:22.151216Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2604.09408"},"observation_digest":"sha256:f668a4347a492f375d3fa4729deb9d157aadfa7528c626b594723a3e496c3b8d","observation_id":"5eab2639-01f2-41c0-9071-b72e9123a2fd","resolution":{"observed_at":"2026-05-12T13:48:53.849224Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2604.11535","last_updated":"2026-05-07T11:10:15Z","snapshot_observed_at":"2026-07-06T22:59:54.573362Z","submitted_at":"2026-04-13T14:32:08Z","title":"Problem Reductions at Scale: Agentic Integration of Computationally Hard Problems","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-10T16:07:35.395872Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2604.11535"},"observation_digest":"sha256:a68c4725bff7dddb895cacef5c3d40127920b7458ce785585ef6e326a59ee112","observation_id":"8a5df470-1c62-4ded-8643-a3592547b717","resolution":{"observed_at":"2026-05-12T13:48:53.849224Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2604.12147","last_updated":"2026-04-28T15:58:17Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-13T23:54:55Z","title":"Evaluating Plan Compliance in Autonomous Programming Agents","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-10T14:52:51.349446Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2604.12147"},"observation_digest":"sha256:39b8c65b7a181bfbd92b2ce021aa69cdf02b89efa856f13f3f6f294618da1662","observation_id":"2f0d715d-d53d-4fd3-8e5d-d106104f7ea0","resolution":{"observed_at":"2026-05-12T13:48:53.849224Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2604.14709","last_updated":"2026-05-05T12:56:34Z","snapshot_observed_at":"2026-07-06T23:02:27.141640Z","submitted_at":"2026-04-16T07:19:34Z","title":"HWE-Bench: Benchmarking LLM Agents on Real-World Hardware Bug Repair Tasks","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-10T11:27:11.257817Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2604.14709"},"observation_digest":"sha256:cc146eeb7811b0ad5a57c4917bb33cdddbe6bb958ace677b4da91ec912af59fc","observation_id":"dd6c239f-0d6f-4699-bbdc-5a86a72e6a27","resolution":{"observed_at":"2026-05-12T13:48:53.849224Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2604.17308","last_updated":"2026-04-19T07:51:46Z","snapshot_observed_at":"2026-07-06T23:04:28.260463Z","submitted_at":"2026-04-19T07:51:46Z","title":"SkillFlow:Benchmarking Lifelong Skill Discovery and Evolution for Autonomous Agents","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-10T06:13:32.434201Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2604.17308"},"observation_digest":"sha256:15cec5cac3c905b4e0e4f1537e1337dadee0a8c37c3eeadc38c32ffb3cefc5d1","observation_id":"c4af8321-575e-4797-8840-69e3b113f30e","resolution":{"observed_at":"2026-05-12T13:48:53.849224Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2604.23065","last_updated":"2026-07-01T23:54:28Z","snapshot_observed_at":"2026-07-06T23:09:19.041332Z","submitted_at":"2026-04-24T23:28:38Z","title":"What Should Frontier AI Developers Disclose About Internal Deployments?","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-05-08T09:33:47.869030Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2604.23065"},"observation_digest":"sha256:3db51ebca20ee8b723330c8c2741ba2b80cc001d3999040596e64e5dd2ecc5df","observation_id":"f8bd41b9-d737-4592-9b4c-5503b1b9a1b9","resolution":{"observed_at":"2026-05-12T13:48:53.849224Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2604.23822","last_updated":"2026-06-29T08:22:31Z","snapshot_observed_at":"2026-08-03T08:04:04.856532Z","submitted_at":"2026-04-26T17:59:06Z","title":"KISS Sorcar: A Stupidly-Simple General-Purpose and Software Engineering AI Assistant","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-08T05:57:20.823210Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2604.23822"},"observation_digest":"sha256:73689c8b45e7237ae69e9f2cbc10b8b19725757b6da1393889961b30a90fc595","observation_id":"69b70e61-fb90-4039-96e5-803f5cd20d3f","resolution":{"observed_at":"2026-05-12T13:48:53.849224Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2604.23822","last_updated":"2026-06-29T08:22:31Z","snapshot_observed_at":"2026-08-03T08:04:04.856532Z","submitted_at":"2026-04-26T17:59:06Z","title":"KISS Sorcar: A Stupidly-Simple General-Purpose and Software Engineering AI Assistant","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-01T09:03:16.808596Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2604.23822"},"observation_digest":"sha256:d77467aad1cf4f5cb1b163689e7858440478bfc821fceb03995924e64e6e5c9f","observation_id":"d410cc8c-814f-4668-841f-cd6609322f23","resolution":{"observed_at":"2026-07-01T09:05:36.451754Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2604.24966","last_updated":"2026-04-27T20:07:09Z","snapshot_observed_at":"2026-07-06T23:10:52.763251Z","submitted_at":"2026-04-27T20:07:09Z","title":"Risk Reporting for Developers' Internal AI Model Use","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-07T17:47:21.321820Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2604.24966"},"observation_digest":"sha256:31add50b40f85a37ee3b4f28f7d46d383dfac5b7fb6e7bfb41c4dc8c34e7d3e1","observation_id":"9cd76b4c-6bd2-4e90-bdea-b5b53fe87986","resolution":{"observed_at":"2026-05-12T13:48:53.849224Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2604.25880","last_updated":"2026-04-28T17:21:46Z","snapshot_observed_at":"2026-08-02T09:47:12.488290Z","submitted_at":"2026-04-28T17:21:46Z","title":"From Threads to Trajectories: A Multi-LLM Pipeline for Community Knowledge Extraction from GitHub Issue Discussions","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-07T15:51:47.520468Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2604.25880"},"observation_digest":"sha256:602dd713435fe13810356da292b331fffcd103a554727447a6fdf712954193e0","observation_id":"8d381dd1-bc04-424a-b780-bf6687025065","resolution":{"observed_at":"2026-05-12T13:48:53.849224Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2604.26469","last_updated":"2026-05-04T13:26:48Z","snapshot_observed_at":"2026-07-06T23:12:06.349884Z","submitted_at":"2026-04-29T09:26:13Z","title":"An Empirical Study of Speculative Decoding on Software Engineering Tasks","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-07T13:35:40.422686Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2604.26469"},"observation_digest":"sha256:15eb10a46300d7a527208f9a7a98aa774c10ec32039b584d7ed4f4a5da23acdb","observation_id":"5dcb735e-0ec5-4e3c-b9af-e2542d42a03d","resolution":{"observed_at":"2026-05-12T13:48:53.849224Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.02244","last_updated":"2026-05-04T05:37:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-05-04T05:37:36Z","title":"The Conversations Beneath the Code: Triadic Data for Long-Horizon Software Engineering Agents","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-05-08T18:30:21.856024Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.02244"},"observation_digest":"sha256:65504e72a47162344e9bf38e3361be799edd7d5583d54173074e4f40ca97bee1","observation_id":"5a413685-774a-4426-aa15-6a61bee37eb5","resolution":{"observed_at":"2026-05-12T13:48:53.849224Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.03195","last_updated":"2026-07-01T17:09:23Z","snapshot_observed_at":"2026-07-12T17:42:50.872609Z","submitted_at":"2026-05-04T22:24:24Z","title":"Terminus-4B: Can a Smaller Model Replace Frontier LLMs at Agentic Execution Tasks?","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-08T18:09:12.419652Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.03195"},"observation_digest":"sha256:e40950dd8106627903b9eb03b6a7e15cd5bd8a6a058f3e300a141099b18bbd67","observation_id":"5226126a-18da-4d8e-827e-b9eaa1bf7171","resolution":{"observed_at":"2026-05-12T13:48:53.849224Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-07-12T17:42:51.275272Z","title":"Swe-bench pro: Can ai agents solve long-horizon software engineering tasks?","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.03195","last_updated":"2026-07-01T17:09:23Z","snapshot_observed_at":"2026-07-12T17:42:50.872609Z","submitted_at":"2026-05-04T22:24:24Z","title":"Terminus-4B: Can a Smaller Model Replace Frontier LLMs at Agentic Execution Tasks?","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-07-12T17:42:51.275272Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.03195"},"observation_digest":"sha256:993e45bd8fb0798f46cdd390fbc700eb94f3b7b91202c2f3a02a67e960b935c7","observation_id":"95b77cf5-352a-4122-95bd-7aadd0001c70","resolution":{"observed_at":"2026-07-12T17:42:51.275272Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.03546","last_updated":"2026-05-05T09:17:02Z","snapshot_observed_at":"2026-07-29T17:17:07.530182Z","submitted_at":"2026-05-05T09:17:02Z","title":"ProgramBench: Can Language Models Rebuild Programs From Scratch?","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-07T16:02:16.598404Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.03546"},"observation_digest":"sha256:2971f5163022eb35d42b83cbd19343a0ef3122bfd1d5a22f6eb4e7ecf6fa8cf5","observation_id":"3ebf38f3-0b04-491a-b912-460c64803631","resolution":{"observed_at":"2026-05-12T13:48:53.849224Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.04320","last_updated":"2026-05-07T19:11:05Z","snapshot_observed_at":"2026-07-06T23:17:09.246351Z","submitted_at":"2026-05-05T21:49:52Z","title":"Reproduction Test Generation for Java SWE Issues","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-08T16:56:33.346935Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.04320"},"observation_digest":"sha256:5eb45adc069bfc6a95ee7a4e7dfcab65255d422cc99832350154efb0f6e2112f","observation_id":"2252a6bc-f4b8-4c1d-8e49-8054c3554788","resolution":{"observed_at":"2026-05-12T13:48:53.849224Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.04320","last_updated":"2026-05-07T19:11:05Z","snapshot_observed_at":"2026-07-06T23:17:09.246351Z","submitted_at":"2026-05-05T21:49:52Z","title":"Reproduction Test Generation for Java SWE Issues","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-11T00:48:28.794195Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.04320"},"observation_digest":"sha256:ad9dbc6e9da6bb5af8898afdaaeffb7c525ad589c1094df9fffe424ab226df6f","observation_id":"f5e6106c-2853-4b7c-8a7c-7c119cba2f7c","resolution":{"observed_at":"2026-05-12T13:48:53.849224Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.06125","last_updated":"2026-05-07T12:31:09Z","snapshot_observed_at":"2026-07-06T23:18:36.537416Z","submitted_at":"2026-05-07T12:31:09Z","title":"Breaking, Stale, or Missing? Benchmarking Coding Agents on Project-Level Test Evolution","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-08T09:04:06.347354Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.06125"},"observation_digest":"sha256:b6c8a66dcc9476f6e4eb23299943edfe0c4f49db5cdc667762b0a6f7cf83bf02","observation_id":"e5097a72-fd71-451d-b850-25b82c93f511","resolution":{"observed_at":"2026-05-12T13:48:53.849224Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.06445","last_updated":"2026-05-07T15:44:40Z","snapshot_observed_at":"2026-08-06T12:31:28.382494Z","submitted_at":"2026-05-07T15:44:40Z","title":"Constraint Decay: The Fragility of LLM Agents in Backend Code Generation","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-05-08T08:48:36.848425Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.06445"},"observation_digest":"sha256:abf5d52393859f47657e9502d92d567373ad1ec08afa60f4abfda240a6679b13","observation_id":"f7cafb1d-2016-4127-9ca4-ea4caad9df08","resolution":{"observed_at":"2026-05-12T13:48:53.849224Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.07937","last_updated":"2026-05-08T16:08:03Z","snapshot_observed_at":"2026-08-04T22:40:32.494365Z","submitted_at":"2026-05-08T16:08:03Z","title":"Ask Early, Ask Late, Ask Right: When Does Clarification Timing Matter for Long-Horizon Agents?","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-05-11T03:24:10.130356Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.07937"},"observation_digest":"sha256:f387d30d44fb2e304b8b8b8d09fb09bc1a0a0ac304fc9fb682ed587e1d6c14dc","observation_id":"54fcd544-adc8-4f04-942c-471ea125fde7","resolution":{"observed_at":"2026-05-12T13:48:53.849224Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.08366","last_updated":"2026-05-08T18:21:44Z","snapshot_observed_at":"2026-07-06T23:20:38.385189Z","submitted_at":"2026-05-08T18:21:44Z","title":"SWE Atlas: Benchmarking Coding Agents Beyond Issue Resolution","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-12T02:28:07.557119Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.08366"},"observation_digest":"sha256:1713172d2f135aaa87b0869542704b4529ecdb83ffe1f08cdeca08bae82974e2","observation_id":"77150a33-e2b1-46f7-813a-1d98e59558ff","resolution":{"observed_at":"2026-05-12T13:48:53.849224Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.09252","last_updated":"2026-05-20T19:57:19Z","snapshot_observed_at":"2026-08-02T16:03:49.427921Z","submitted_at":"2026-05-10T01:37:40Z","title":"LLM Agents Already Know When to Call Tools -- Even Without Reasoning","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-05-12T02:44:00.329276Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.09252"},"observation_digest":"sha256:8f49ee8462b9aaa6dbe01a73afed031268c418bc7d3c49d8a0426a41801e7293","observation_id":"a504efaa-d9c0-4462-a00c-eda78f883fed","resolution":{"observed_at":"2026-05-12T13:48:53.849224Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.09252","last_updated":"2026-05-20T19:57:19Z","snapshot_observed_at":"2026-08-02T16:03:49.427921Z","submitted_at":"2026-05-10T01:37:40Z","title":"LLM Agents Already Know When to Call Tools -- Even Without Reasoning","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-05-22T10:55:14.037226Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.09252"},"observation_digest":"sha256:1f7f50d0d1071751ff2cc26cf97a449b4fc7316faa0baab4b5ae7e30e5c4d395","observation_id":"e8ef8e8c-9e34-4861-b2a7-a19be5da0d1f","resolution":{"observed_at":"2026-05-22T10:56:25.958471Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.09636","last_updated":"2026-05-10T16:25:43Z","snapshot_observed_at":"2026-07-30T14:07:22.496707Z","submitted_at":"2026-05-10T16:25:43Z","title":"PDEAgent-Bench: A Multi-Metric, Multi-Library Benchmark for PDE Solver Generation","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-12T02:36:40.696567Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.09636"},"observation_digest":"sha256:72cb0ea3e12c1ab37c05fa10ce1686286fba4abd8cfee7e4887225515a584c08","observation_id":"a84ee52e-36b6-4a4d-be4e-32d46abc0950","resolution":{"observed_at":"2026-05-12T13:48:53.849224Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.10754","last_updated":"2026-05-11T15:53:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-05-11T15:53:54Z","title":"The Agent Use of Agent Beings: Agent Cybernetics Is the Missing Science of Foundation Agents","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-12T05:03:58.419364Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.10754"},"observation_digest":"sha256:61e8c8505e2d99081e37d8fcb7301cfddf4a8aca398a0ad4503e41fa648bdac6","observation_id":"d83f093e-718b-40f0-bcea-3fed8c8fd736","resolution":{"observed_at":"2026-05-12T13:48:53.849224Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-02T02:31:32.394072Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:1b3918dafa44b986d074b81b319580c81f1fb96bb3ba50d5e60fb5b4da59a7d3","observation_id":"eb042689-a284-479a-8f12-f6a3a55d4fd3","resolution":{"observed_at":"2026-05-14T20:32:56.742365Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.12913","last_updated":"2026-05-13T02:40:28Z","snapshot_observed_at":"2026-07-06T23:24:32.412966Z","submitted_at":"2026-05-13T02:40:28Z","title":"Revisiting DAgger in the Era of LLM-Agents","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-14T19:56:06.762156Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.12913"},"observation_digest":"sha256:28436455da1f1e4f86a25be0d7ed2ad976c82e038d42dbfda50e549fd533b8ac","observation_id":"c8ade44b-5959-4279-a90b-bffac6b7c730","resolution":{"observed_at":"2026-05-14T19:57:53.403116Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.13139","last_updated":"2026-05-13T08:05:16Z","snapshot_observed_at":"2026-07-06T23:24:47.309044Z","submitted_at":"2026-05-13T08:05:16Z","title":"SWE-Cycle: Benchmarking Code Agents across the Complete Issue Resolution Cycle","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-14T18:34:39.997353Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.13139"},"observation_digest":"sha256:b82b75502237686dc999f7f95d644a989b7d8041e94675f94e7e91aa649b9468","observation_id":"06d7c0a3-9c97-4644-b93b-5a6fc3091519","resolution":{"observed_at":"2026-05-14T18:37:35.647810Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.14563","last_updated":"2026-07-10T14:40:19Z","snapshot_observed_at":"2026-08-03T09:09:17.135940Z","submitted_at":"2026-05-14T08:35:20Z","title":"Remember Your Trace: Memory-Guided Long-Horizon Agentic Framework for Consistent and Hierarchical Repository-Level Code Documentation","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-15T01:38:15.112910Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.14563"},"observation_digest":"sha256:ce7eb8b20649dfbb5ab3bc8f7400c69d13f93d7f1dd53881a7b2d528ab178ca0","observation_id":"b59972cb-0de7-4587-8e8a-2fac2b9f984d","resolution":{"observed_at":"2026-05-15T01:38:27.489459Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-07-13T07:56:37.112844Z","title":"Swe-bench pro: Can ai agents solve long-horizon software engineering tasks?arXiv preprint arXiv:2509.16941, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.14563","last_updated":"2026-07-10T14:40:19Z","snapshot_observed_at":"2026-08-03T09:09:17.135940Z","submitted_at":"2026-05-14T08:35:20Z","title":"Remember Your Trace: Memory-Guided Long-Horizon Agentic Framework for Consistent and Hierarchical Repository-Level Code Documentation","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-07-13T07:56:37.112844Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.14563"},"observation_digest":"sha256:5411b31cf5bed7c81939b61bad5ca491cb13b77b96e42f7707ec537b3c069696","observation_id":"0ab82e84-cc86-4718-b255-fdc308fd9099","resolution":{"observed_at":"2026-07-13T07:56:37.112844Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.17526","last_updated":"2026-05-17T16:15:56Z","snapshot_observed_at":"2026-08-04T14:19:36.942893Z","submitted_at":"2026-05-17T16:15:56Z","title":"SaaSBench: Exploring the Boundaries of Coding Agents in Long-Horizon Enterprise SaaS Engineering","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-19T22:43:13.251812Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.17526"},"observation_digest":"sha256:13f926157014b71a04bda0d8171ad68edcf1596d18e8defea604c7a59d77108c","observation_id":"7d198de4-fee1-4182-9719-31488dd2f51d","resolution":{"observed_at":"2026-05-19T22:43:14.940231Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.18401","last_updated":"2026-05-18T13:44:19Z","snapshot_observed_at":"2026-07-06T23:29:20.279470Z","submitted_at":"2026-05-18T13:44:19Z","title":"SkillsVote: Lifecycle Governance of Agent Skills from Collection, Recommendation to Evolution","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-20T11:40:45.397038Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.18401"},"observation_digest":"sha256:553fd7a7e1fb5c8afef01261339308e0f9dc265e3079641af4f96ea57891e1c2","observation_id":"9a80e3eb-b345-4428-ab23-02f0038ead57","resolution":{"observed_at":"2026-05-20T11:43:15.136412Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.18661","last_updated":"2026-07-20T17:24:03Z","snapshot_observed_at":"2026-08-02T13:43:29.187658Z","submitted_at":"2026-05-18T17:08:26Z","title":"AI for Auto-Research: Roadmap & User Guide","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-20T10:30:50.256635Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.18661"},"observation_digest":"sha256:efefb310c3816145fb1adca62c04e073e5fc01cd9d2e6f88eaa0e661e3ceef31","observation_id":"d7bc1a6d-a13d-4e66-8202-8bc9bd28b69f","resolution":{"observed_at":"2026-05-20T10:33:12.743474Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-02T13:43:32.945342Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.18661","last_updated":"2026-07-20T17:24:03Z","snapshot_observed_at":"2026-08-02T13:43:29.187658Z","submitted_at":"2026-05-18T17:08:26Z","title":"AI for Auto-Research: Roadmap & User Guide","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-02T13:43:32.945342Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.18661"},"observation_digest":"sha256:9c3144ac4a1a998f4475ee21975c5a880014baa975faa183b18cfe3a66851bca","observation_id":"311134f0-7ca3-4a19-90ff-6b762e1931c3","resolution":{"observed_at":"2026-08-02T13:43:32.945342Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.20520","last_updated":"2026-05-19T21:42:32Z","snapshot_observed_at":"2026-08-03T02:30:24.579386Z","submitted_at":"2026-05-19T21:42:32Z","title":"Open-World Evaluations for Measuring Frontier AI Capabilities","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-05-21T06:38:51.427985Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.20520"},"observation_digest":"sha256:452c319983693e28ad1938035d6593d7250a48d40799f35c45f33f52db0c34e9","observation_id":"92e16258-0172-4c27-9d02-66cab342ed7b","resolution":{"observed_at":"2026-05-21T06:39:43.728663Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.21100","last_updated":"2026-05-20T12:28:51Z","snapshot_observed_at":"2026-08-02T15:55:38.422631Z","submitted_at":"2026-05-20T12:28:51Z","title":"NanoCP: Request-Level Dynamic Context Parallelism for Data-Expert Parallel Decoding","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-21T01:51:38.671365Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.21100"},"observation_digest":"sha256:66952ee0b338e4471dd05ea62175d8ab72cf5f208171a529f52f886b3e4e9717","observation_id":"0f26fd98-371a-42f8-9caf-d8ac484d49e4","resolution":{"observed_at":"2026-05-21T01:53:54.613591Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.21347","last_updated":"2026-06-04T23:27:42Z","snapshot_observed_at":"2026-07-06T23:31:46.877766Z","submitted_at":"2026-05-20T16:13:53Z","title":"Insights Generator: Systematic Corpus-Level Trace Diagnostics for LLM Agents","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-21T04:20:43.780849Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.21347"},"observation_digest":"sha256:74d3ff8b4a18a348e11e8553391edf1595d9d9bb53ecf45fb98170fa3d4ebb48","observation_id":"5bee8a10-af87-431e-842f-9cf15f3f0aa1","resolution":{"observed_at":"2026-05-21T04:23:57.583837Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.21347","last_updated":"2026-06-04T23:27:42Z","snapshot_observed_at":"2026-07-06T23:31:46.877766Z","submitted_at":"2026-05-20T16:13:53Z","title":"Insights Generator: Systematic Corpus-Level Trace Diagnostics for LLM Agents","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-22T09:46:26.124683Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.21347"},"observation_digest":"sha256:8f0a03b5531507c5d91d7f711327e671891826a49859fe17df5a2a282c1effe3","observation_id":"b89c8d84-bfd6-4425-9fde-090fade0186f","resolution":{"observed_at":"2026-05-22T09:51:21.923553Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.21347","last_updated":"2026-06-04T23:27:42Z","snapshot_observed_at":"2026-07-06T23:31:46.877766Z","submitted_at":"2026-05-20T16:13:53Z","title":"Insights Generator: Systematic Corpus-Level Trace Diagnostics for LLM Agents","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-30T17:31:25.538977Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.21347"},"observation_digest":"sha256:d2d5998b112ba95142354d48cd9b8a9ace72972bfaa6eaebfa5c7d73de725b5a","observation_id":"a92ef363-b45f-4990-b02c-2b60d60f0cff","resolution":{"observed_at":"2026-06-30T17:34:57.654428Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.21384","last_updated":"2026-05-20T16:41:51Z","snapshot_observed_at":"2026-08-03T04:17:36.867521Z","submitted_at":"2026-05-20T16:41:51Z","title":"SpecBench: Measuring Reward Hacking in Long-Horizon Coding Agents","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-05-21T03:07:18.501526Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.21384"},"observation_digest":"sha256:54288c18710654efa7d04f4cc4aca1eca983798e290b518a8a206bce17f392d2","observation_id":"f0114c10-1b66-4e8e-8580-10434f5800a5","resolution":{"observed_at":"2026-05-21T03:09:28.093912Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.21463","last_updated":"2026-05-20T17:51:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-05-20T17:51:05Z","title":"Mem-$\\pi$: Adaptive Memory through Learning When and What to Generate","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-21T04:27:25.041652Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.21463"},"observation_digest":"sha256:a9ff2cd319d551738a215dfd07047de048b6e8544429c6001eabf0205c2f2e46","observation_id":"90a487ad-326a-4460-8376-7407687d6d3c","resolution":{"observed_at":"2026-05-21T04:29:34.576114Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.21996","last_updated":"2026-05-21T04:54:55Z","snapshot_observed_at":"2026-07-06T23:32:25.286443Z","submitted_at":"2026-05-21T04:54:55Z","title":"From Patches to Trajectories: Privileged Process Supervision for Software-Engineering Agents","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-22T05:13:26.057531Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.21996"},"observation_digest":"sha256:14cf7cbd327a3f92c4b45172ba803ae380afca4955656557ca0830a3cd8b9850","observation_id":"c10373ab-a14e-46ad-9db8-3d85178d6f65","resolution":{"observed_at":"2026-05-22T05:14:37.631743Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.23262","last_updated":"2026-05-22T06:03:01Z","snapshot_observed_at":"2026-08-02T00:22:40.506306Z","submitted_at":"2026-05-22T06:03:01Z","title":"Design and Report Benchmarks for Knowledge Work","version":1},"reference_index":113,"source":"arxiv_source","source_observed_at":"2026-05-25T04:39:14.319133Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.23262"},"observation_digest":"sha256:06f1d76da25122d4a36c14052e5de7b06f5de90f9041d03fea36d79413dd608e","observation_id":"05de791f-d537-42cb-b540-d2f9f442d50c","resolution":{"observed_at":"2026-05-25T04:40:22.646987Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.23950","last_updated":"2026-05-07T15:24:59Z","snapshot_observed_at":"2026-08-03T13:44:55.519841Z","submitted_at":"2026-05-07T15:24:59Z","title":"Stop Comparing LLM Agents Without Disclosing the Harness","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-30T23:15:37.160073Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.23950"},"observation_digest":"sha256:a2bdd4e1db6d36070272e29aa9bef3219232ecba99344771c2811042aafd4b5e","observation_id":"b5985207-77b9-45da-80a5-780d2bacb82c","resolution":{"observed_at":"2026-07-01T13:25:45.870677Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.26177","last_updated":"2026-05-25T06:26:43Z","snapshot_observed_at":"2026-07-06T23:36:06.135765Z","submitted_at":"2026-05-25T06:26:43Z","title":"RepoMirage: Probing Repository Context Reasoning in Code Agents with Perturbations","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-29T20:53:30.382870Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.26177"},"observation_digest":"sha256:f8baf2ea52cd8f438e346529e8c9bc5584fd2b1b116d4b989aa9b2ce4fdff432","observation_id":"fc23a8af-ad45-4bc1-8394-9afb0e7626ed","resolution":{"observed_at":"2026-06-29T20:53:57.818067Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.26297","last_updated":"2026-05-25T19:45:21Z","snapshot_observed_at":"2026-08-03T04:38:08.316378Z","submitted_at":"2026-05-25T19:45:21Z","title":"Agentic AI Workload Characteristics","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-29T20:11:50.722787Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.26297"},"observation_digest":"sha256:018a0df2481e701cacabaae67f8d6ed426e07323debd2aff216e5232872a41d3","observation_id":"3d95fc5b-f528-4013-b57c-824d67e1275a","resolution":{"observed_at":"2026-06-29T20:13:58.835453Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.26548","last_updated":"2026-07-20T16:24:12Z","snapshot_observed_at":"2026-08-04T21:47:01.026446Z","submitted_at":"2026-05-26T04:59:49Z","title":"SEC-bench Pro: Can Language Models Solve Long-Horizon Software Security Tasks?","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-29T17:29:36.006340Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.26548"},"observation_digest":"sha256:7e9616b54d4a55a6f820ccdf05c052ab3a068926475ff42743a1e63d9025a41e","observation_id":"74362d7c-e805-4960-93e9-4499ec8a3d65","resolution":{"observed_at":"2026-06-29T17:33:45.250290Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-02T13:10:39.062654Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?arXiv preprint arXiv:2509.16941,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.26548","last_updated":"2026-07-20T16:24:12Z","snapshot_observed_at":"2026-08-04T21:47:01.026446Z","submitted_at":"2026-05-26T04:59:49Z","title":"SEC-bench Pro: Can Language Models Solve Long-Horizon Software Security Tasks?","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-02T13:10:39.062654Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.26548"},"observation_digest":"sha256:3f79ef626aec1add8b253c87e4123cde747bece69d2ef88e99332b4bd64bc9ad","observation_id":"6cd2488a-f694-4ced-898c-f6f4d4c8cbf8","resolution":{"observed_at":"2026-08-02T13:10:39.062654Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.26563","last_updated":"2026-07-28T05:47:02Z","snapshot_observed_at":"2026-08-02T13:09:12.493449Z","submitted_at":"2026-05-26T05:24:37Z","title":"TrajAudit: Automated Failure Diagnosis for Agentic Coding Systems","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-29T16:13:30.977123Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.26563"},"observation_digest":"sha256:81b17e0202fe1e800892b9cfc721c828c819f0110e9ea5bcfb417f2d97ac0a26","observation_id":"b6315f07-7902-4aa9-8442-d029cbd258d9","resolution":{"observed_at":"2026-06-29T16:13:35.844714Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-02T13:09:14.215805Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.26563","last_updated":"2026-07-28T05:47:02Z","snapshot_observed_at":"2026-08-02T13:09:12.493449Z","submitted_at":"2026-05-26T05:24:37Z","title":"TrajAudit: Automated Failure Diagnosis for Agentic Coding Systems","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-02T13:09:14.215805Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.26563"},"observation_digest":"sha256:f945e17ac2b6b1b7dc9bfbf43b40199730603e4a0087bbce8ef0bb5d355653a1","observation_id":"46ca2011-88b9-40a1-9473-b1e2431d08b9","resolution":{"observed_at":"2026-08-02T13:09:14.215805Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.27492","last_updated":"2026-05-26T16:28:10Z","snapshot_observed_at":"2026-07-31T21:45:32.197960Z","submitted_at":"2026-05-26T16:28:10Z","title":"Benchmarks are Not Enough: RAMP for Runtime Assessing of Agentic Models in Production Systems","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-29T15:32:21.737028Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.27492"},"observation_digest":"sha256:6bfb22bfeeac226fa58b6df463f06008e9349d97e744ca1727416749ad9bde21","observation_id":"38d2dc92-3a74-47c3-b03e-2a31eefbb616","resolution":{"observed_at":"2026-06-29T15:33:32.782352Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.29653","last_updated":"2026-05-28T09:16:22Z","snapshot_observed_at":"2026-07-06T23:39:01.234601Z","submitted_at":"2026-05-28T09:16:22Z","title":"PTCG-Bench: Can LLM Agents Master Pok\\'emon Trading Card Game?","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-29T07:21:49.763994Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.29653"},"observation_digest":"sha256:07a5873cf808aa7af6a89200e4c00fed81275010125510f1cdf690d75d90bf24","observation_id":"51a6c18d-b57e-404d-ab47-1d6f4aac611c","resolution":{"observed_at":"2026-06-29T07:23:12.590793Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.29790","last_updated":"2026-05-28T11:40:16Z","snapshot_observed_at":"2026-08-06T19:17:26.541233Z","submitted_at":"2026-05-28T11:40:16Z","title":"Evolve as a Team: Collaborative Self-Evolution for LLM-based Multi-Agent Systems","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-29T00:05:31.780655Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.29790"},"observation_digest":"sha256:e754c9ffedc5e2b06e91e128b6f9b80a20ddaddbe26806704818a0c2afabce8a","observation_id":"037da2a1-4226-4704-b11d-4c336c4f9fc1","resolution":{"observed_at":"2026-06-29T00:12:50.174304Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2605.30907","last_updated":"2026-05-29T06:43:23Z","snapshot_observed_at":"2026-07-06T23:40:07.833692Z","submitted_at":"2026-05-29T06:43:23Z","title":"BlueFin: Benchmarking LLM Agents on Financial Spreadsheets","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-28T21:35:45.197423Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2605.30907"},"observation_digest":"sha256:2549861fe3f15cfaecedc5042f559d4a60d9785155a167280905470845dfc474","observation_id":"46bad182-e56a-4676-ad27-cfd6422c3c78","resolution":{"observed_at":"2026-07-01T20:06:12.323633Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2606.00866","last_updated":"2026-05-30T19:44:25Z","snapshot_observed_at":"2026-08-06T18:53:41.253747Z","submitted_at":"2026-05-30T19:44:25Z","title":"Idleness is Relative: Exploiting Tool-Call Idle Windows for Offloading in Agentic Systems with MORI","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-28T17:30:56.324289Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2606.00866"},"observation_digest":"sha256:bab5398650fe7e6afbe184fd3797b2ffd50a816a99fd8c5021d4ea58db174a68","observation_id":"6bceb751-5100-452e-a60a-4ac3614bccdb","resolution":{"observed_at":"2026-07-01T21:06:13.588462Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2606.03103","last_updated":"2026-06-02T03:42:34Z","snapshot_observed_at":"2026-07-30T08:46:30.981727Z","submitted_at":"2026-06-02T03:42:34Z","title":"DeskCraft: Benchmarking Desktop Agents on Professional Workflows and Human-in-the-Loop Collaboration","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-28T10:33:20.920760Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2606.03103"},"observation_digest":"sha256:4595f36035934a4abcac07da695bd6059e41373e64fd689b86ee9c310bed0483","observation_id":"3c6ac21d-2210-41c6-a001-7b6bae2bc7d9","resolution":{"observed_at":"2026-07-02T02:56:28.775459Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2606.05570","last_updated":"2026-06-04T01:42:40Z","snapshot_observed_at":"2026-08-02T22:19:59.617747Z","submitted_at":"2026-06-04T01:42:40Z","title":"TensorBench: Benchmarking Coding Agents on a Compiler-Based Tensor Framework","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:59.002171Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2606.05570"},"observation_digest":"sha256:4f10600ea4c6ff0c4a806c84c4270b237f64b8e53b2f614c8cf72462dfcb06fe","observation_id":"bc911c64-2378-4cb3-8db6-81ffbc66ac0f","resolution":{"observed_at":"2026-07-02T12:36:57.385694Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2606.05661","last_updated":"2026-06-04T03:43:28Z","snapshot_observed_at":"2026-07-06T23:45:37.713710Z","submitted_at":"2026-06-04T03:43:28Z","title":"Continual Learning Bench: Evaluating Frontier AI Systems in Real-World Stateful Environments","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-28T01:45:28.693098Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2606.05661"},"observation_digest":"sha256:2a8e33612ba2bef74bf5af6fb5a12ddcd8df0788ca52bd07f8c0a183a2bc457c","observation_id":"fe56401d-1c7d-4c4b-a87f-6c00174200d9","resolution":{"observed_at":"2026-07-02T12:56:56.839653Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2606.05922","last_updated":"2026-06-10T05:04:53Z","snapshot_observed_at":"2026-08-03T16:23:21.190194Z","submitted_at":"2026-06-04T09:26:00Z","title":"Evolving Agents in the Dark: Retrospective Harness Optimization via Self-Preference","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-28T01:41:32.686349Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2606.05922"},"observation_digest":"sha256:2b2790aabd2ebb24dde1a22e1e9dd43c6bbc45ac01ae8c489b34983a2c2d99b2","observation_id":"ebdbb5be-f447-4acd-ab0e-212b6ac70a40","resolution":{"observed_at":"2026-07-02T12:56:57.375937Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2606.07297","last_updated":"2026-06-05T14:08:27Z","snapshot_observed_at":"2026-07-06T23:46:55.970997Z","submitted_at":"2026-06-05T14:08:27Z","title":"SWE-Explore: Benchmarking How Coding Agents Explore Repositories","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-27T21:21:43.468977Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2606.07297"},"observation_digest":"sha256:02ec8c5c41c076bcad00d9ac944ccd8475037eb72b7150c4dc3b11de3e17686e","observation_id":"7f14fcfe-164d-46dc-a39a-0f60100073ef","resolution":{"observed_at":"2026-07-02T19:47:19.218108Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2606.12191","last_updated":"2026-06-10T15:15:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-06-10T15:15:01Z","title":"Agentic Environment Engineering for Large Language Models: A Survey of Environment Modeling, Synthesis, Evaluation, and Application","version":1},"reference_index":151,"source":"pdf_text","source_observed_at":"2026-06-27T09:46:30.702256Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2606.12191"},"observation_digest":"sha256:25391a1cea749eeadb2b615182fa19d85b5ad8cc500bc25a40ef87f01d70d2c9","observation_id":"0656c1f3-66af-4007-a018-7be7e408ce6f","resolution":{"observed_at":"2026-06-27T09:50:48.437287Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2606.12344","last_updated":"2026-06-10T17:16:23Z","snapshot_observed_at":"2026-08-02T03:07:28.486057Z","submitted_at":"2026-06-10T17:16:23Z","title":"Claw-SWE-Bench: A Benchmark for Evaluating OpenClaw-style Agent Harnesses on Coding Tasks","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-27T10:39:51.630229Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2606.12344"},"observation_digest":"sha256:86be384561aa79c6058ffef17839a3a03318103556f780c3cc18adb136a44b1f","observation_id":"54d18632-e300-4c7e-a44d-55bdc2d6c79f","resolution":{"observed_at":"2026-07-03T08:57:47.632014Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2606.13608","last_updated":"2026-06-11T17:23:54Z","snapshot_observed_at":"2026-07-06T23:52:18.996096Z","submitted_at":"2026-06-11T17:23:54Z","title":"AgentBeats: Agentifying Agent Assessment for Openness, Standardization, and Reproducibility","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-27T06:41:41.799596Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2606.13608"},"observation_digest":"sha256:2dd1fb7ae4d8f3c0b44c778cdae86b59202771decfbf037d5a3254654ca0429f","observation_id":"eeb84929-671d-412e-88bc-453e6662a8e7","resolution":{"observed_at":"2026-06-27T07:10:41.772430Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2606.15079","last_updated":"2026-06-13T03:21:49Z","snapshot_observed_at":"2026-07-06T23:52:28.557859Z","submitted_at":"2026-06-13T03:21:49Z","title":"Ling and Ring 2.6 Technical Report: Efficient and Instant Agentic Intelligence at Trillion-Parameter Scale","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-07-02T22:10:59.568675Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2606.15079"},"observation_digest":"sha256:0c1a23df210272934af2809398f9beb35e01ee7f7692706c451932f311fd2f7f","observation_id":"b428d440-dbb6-4f23-94ef-820c10d84236","resolution":{"observed_at":"2026-07-02T22:17:25.678621Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2606.17454","last_updated":"2026-06-17T04:51:06Z","snapshot_observed_at":"2026-08-02T05:33:09.695431Z","submitted_at":"2026-06-16T03:17:03Z","title":"Dissecting model behavior through agent trajectories","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-27T01:27:39.812496Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2606.17454"},"observation_digest":"sha256:0b3b335cc81d4a42aa0050f602f9eed38b35577234525ad08ec10cbca60dffe4","observation_id":"4de31687-758a-479c-8de5-17dd8a57e262","resolution":{"observed_at":"2026-07-03T20:18:56.927589Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2606.17799","last_updated":"2026-07-18T22:14:13Z","snapshot_observed_at":"2026-08-02T11:05:59.500924Z","submitted_at":"2026-06-16T11:21:01Z","title":"Position: Coding Benchmarks Are Misaligned with Agentic Software Engineering","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-26T23:48:58.497927Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2606.17799"},"observation_digest":"sha256:83ab53ba3cff809ed7191cc379d1a093c74597a022b46d9c4fe21245319d6d0f","observation_id":"08baac0d-7a43-4972-a0ba-a83fc0b01753","resolution":{"observed_at":"2026-07-03T22:08:58.889649Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-02T11:06:03.064744Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.17799","last_updated":"2026-07-18T22:14:13Z","snapshot_observed_at":"2026-08-02T11:05:59.500924Z","submitted_at":"2026-06-16T11:21:01Z","title":"Position: Coding Benchmarks Are Misaligned with Agentic Software Engineering","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-02T11:06:03.064744Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2606.17799"},"observation_digest":"sha256:09f8226248190f882fef2bcc1a97cf7a3a3435649952d339c41a95d259931904","observation_id":"3a91eb18-0a35-4798-8dd1-c3ad097ae358","resolution":{"observed_at":"2026-08-02T11:06:03.064744Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2606.17819","last_updated":"2026-06-16T11:46:56Z","snapshot_observed_at":"2026-08-02T04:55:02.510882Z","submitted_at":"2026-06-16T11:46:56Z","title":"A Framework for Evaluating Agentic Skills at Scale","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-06-26T23:42:37.325511Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2606.17819"},"observation_digest":"sha256:57ee3d38540a7fb9c5f7a4c6a81d3069c3cfa20340f7dac03f9fe91c8fb08f32","observation_id":"3a4a2b6b-4888-46f3-913f-78879420a5a9","resolution":{"observed_at":"2026-07-03T22:08:59.894232Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2606.18284","last_updated":"2026-06-10T02:04:29Z","snapshot_observed_at":"2026-07-06T23:53:45.117607Z","submitted_at":"2026-06-10T02:04:29Z","title":"Breaking the Solver Bottleneck: Training Task Generators at the Learnable Frontier","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-06-27T10:36:09.211639Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2606.18284"},"observation_digest":"sha256:da3c61da2a6ef2e7e0acab3d9c100f68768a1c56eb99e1d99f755320e3c01747","observation_id":"fc0d4a20-cd2a-42be-8c3a-c3dd72d3102b","resolution":{"observed_at":"2026-07-03T08:57:48.108196Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2606.19308","last_updated":"2026-06-17T17:31:06Z","snapshot_observed_at":"2026-08-04T23:52:08.389322Z","submitted_at":"2026-06-17T17:31:06Z","title":"Enhancing Decision-Making with Large Language Models through Multi-Agent Fictitious Play","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-26T20:57:49.840546Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2606.19308"},"observation_digest":"sha256:667ada2f451f6f246d5fde6b4587e330101558416e6fd4d6cb76c887b83e814e","observation_id":"90e7165b-426d-408f-9739-52eba309fccd","resolution":{"observed_at":"2026-07-04T00:49:18.510064Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2606.19613","last_updated":"2026-06-17T21:36:09Z","snapshot_observed_at":"2026-08-06T20:08:22.959123Z","submitted_at":"2026-06-17T21:36:09Z","title":"StaminaBench: Stress-Testing Coding Agents over 100 Interaction Turns","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-26T19:47:59.090249Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2606.19613"},"observation_digest":"sha256:903888d0c1e4d9c49122c4ccc244e9d5a1bca9397d07d26ee48cdbd2478f7fcf","observation_id":"68aedaf1-b5a4-4b14-a4ef-e54b0d753528","resolution":{"observed_at":"2026-07-04T02:19:23.944512Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2606.21228","last_updated":"2026-06-23T04:40:39Z","snapshot_observed_at":"2026-08-02T19:24:45.644999Z","submitted_at":"2026-06-19T08:47:40Z","title":"Sakana Fugu Technical Report","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-06-26T14:22:37.596720Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2606.21228"},"observation_digest":"sha256:3400b79c5523f3b2dc30cb7812cc90f2fd15d38423225b0a21c52c4ab018c6a6","observation_id":"97151aee-e408-496f-acda-bd857ebc1ccf","resolution":{"observed_at":"2026-07-04T06:39:36.919614Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2606.21401","last_updated":"2026-06-28T04:34:57Z","snapshot_observed_at":"2026-07-06T23:56:26.805329Z","submitted_at":"2026-06-19T13:11:15Z","title":"SwarmX: Agentic Scheduling for Low-Latency Agentic Systems","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-26T13:10:48.369285Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2606.21401"},"observation_digest":"sha256:553d8d70a1a7e8528b44cb8938c15e7cfe275e459f86a6bc14432c7036006891","observation_id":"4bd2557e-bdf9-4d9c-935d-b65da3acb0d0","resolution":{"observed_at":"2026-07-04T07:39:39.048754Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2606.21401","last_updated":"2026-06-28T04:34:57Z","snapshot_observed_at":"2026-07-06T23:56:26.805329Z","submitted_at":"2026-06-19T13:11:15Z","title":"SwarmX: Agentic Scheduling for Low-Latency Agentic Systems","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-30T10:40:21.294555Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2606.21401"},"observation_digest":"sha256:d54ae87aad49b37960de46b45b03aaceaab2de2cddf33130295e7758e3eb09ce","observation_id":"750246b3-5d42-40d2-b85e-48eda54ea020","resolution":{"observed_at":"2026-06-30T10:44:36.554229Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2606.22000","last_updated":"2026-06-20T11:34:52Z","snapshot_observed_at":"2026-08-03T18:53:35.915270Z","submitted_at":"2026-06-20T11:34:52Z","title":"CFAgentBench: A Reproducible Environment and Benchmark for Autonomous Construction-Finance Agents","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-26T11:50:14.868505Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2606.22000"},"observation_digest":"sha256:8d132dc640ac4e755e45813e7ad18eb5aca150b2355eab1f5e38ad44afc24b2d","observation_id":"e1f62e3e-8627-456f-bd3e-a213102da0cd","resolution":{"observed_at":"2026-07-04T08:19:44.527905Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2606.22417","last_updated":"2026-06-21T10:10:51Z","snapshot_observed_at":"2026-08-03T13:00:47.739114Z","submitted_at":"2026-06-21T10:10:51Z","title":"Code Isn't Memory: A Structural Codebase Index Inside a Coding Agent","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-26T10:59:52.756992Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2606.22417"},"observation_digest":"sha256:40924da9659141d1c2421fd965ca4dd25956441c2b58caa1834befbc72279706","observation_id":"91914b4e-2ab9-4738-a16c-95041df088df","resolution":{"observed_at":"2026-07-04T08:49:41.790205Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":"2509.16941","doi":"10.48550/arxiv.2509.16941","metadata_source":"pith","pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","venue":"cs.SE","work_id":"a561c78a-4b02-4053-a92a-bc5c7c5f6b9b","year":2025},"citing_paper":{"arxiv_id":"2606.23127","last_updated":"2026-06-22T10:14:11Z","snapshot_observed_at":"2026-08-05T10:18:48.889133Z","submitted_at":"2026-06-22T10:14:11Z","title":"Managing Procedural Memory in LLM Agents: Control, Adaptation, and Evaluation","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-06-26T08:39:41.575494Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2606.23127"},"observation_digest":"sha256:b104f214b3819719a99d5c5dcbd556e13176899c1ad799e842180f938f978bc9","observation_id":"5fef82c6-8b56-48dc-8757-99ded1855f77","resolution":{"observed_at":"2026-06-26T08:49:15.248983Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:48.977196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2509.16941/citation-record","integrity":"/paper/2509.16941/integrity","json":"/paper/2509.16941/citation-record.json","paper":"/paper/2509.16941"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2410.06992","last_updated":"2024-10-10T13:13:09Z","snapshot_observed_at":"2026-08-05T03:15:38.088849Z","submitted_at":"2024-10-09T15:38:53Z","title":"SWE-Bench+: Enhanced Coding Benchmark for LLMs","version":2},"cited_work":{"arxiv_id":"2410.06992","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06992","snapshot_observed_at":"2026-07-09T06:06:01.699658Z","title":"ArXiv, abs/2410.06992","venue":"cs.SE","work_id":"a0e2ddd0-ff8a-46d5-a2d8-234dae74cc22","year":2024},"citing_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-12T13:48:53.691192Z"},"links":{"cited_paper":"/paper/2410.06992","citing_paper":"/paper/2509.16941"},"observation_digest":"sha256:e3cb3a052eda0f77219fa0fa59ff25ea6da1bad08447ac1ee0412ebf3ae3ff53","observation_id":"b58572c5-3108-493a-9244-811664d4d9a7","resolution":{"observed_at":"2026-05-12T13:48:53.720917Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2108.07732","last_updated":"2021-08-16T03:57:30Z","snapshot_observed_at":"2026-08-02T19:23:53.535075Z","submitted_at":"2021-08-16T03:57:30Z","title":"Program Synthesis with Large Language Models","version":1},"cited_work":{"arxiv_id":"2108.07732","doi":"10.1007/s11390-025-5518-5","metadata_source":"pith","pith_arxiv_id":"2108.07732","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Program Synthesis with Large Language Models","venue":"cs.PL","work_id":"fd241a05-03b9-4de2-9588-9d77ce176125","year":2021},"citing_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-12T13:48:53.691192Z"},"links":{"cited_paper":"/paper/2108.07732","citing_paper":"/paper/2509.16941"},"observation_digest":"sha256:fe51c4051f2d7e86b2c7c273b83ff3a00027caf17f1ab3ece340d14be5127b73","observation_id":"4fa22eb7-74dd-4d6f-896e-412f97d37a18","resolution":{"observed_at":"2026-05-12T13:48:53.728075Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Brown, B","venue":null,"work_id":"62ffcced-2dcc-44e3-aa5e-6e2a37972c3c","year":1901},"citing_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-12T13:48:53.691192Z"},"links":{"citing_paper":"/paper/2509.16941"},"observation_digest":"sha256:07fbae54d4c8ed95b18640241d3cb45a9776788c7c984ceba74e898c91f992e3","observation_id":"2730ce7e-26c4-4526-ad71-7659dfc36c60","resolution":{"observed_at":"2026-05-12T13:48:53.806530Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2107.03374","last_updated":"2021-07-14T17:16:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-07-07T17:41:24Z","title":"Evaluating Large Language Models Trained on Code","version":2},"cited_work":{"arxiv_id":"2107.03374","doi":"10.48550/arxiv.2107.03374","metadata_source":"pith","pith_arxiv_id":"2107.03374","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating Large Language Models Trained on Code","venue":"cs.LG","work_id":"042493e9-b26f-4b4e-bbde-382072ca9b08","year":2021},"citing_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-12T13:48:53.691192Z"},"links":{"cited_paper":"/paper/2107.03374","citing_paper":"/paper/2509.16941"},"observation_digest":"sha256:aefe259d71c32acd3feb7f4843e162496e6baa91d49845f83ecd0cf9845fd1d0","observation_id":"d00b35bf-a19e-4727-80c4-519562a10538","resolution":{"observed_at":"2026-05-12T13:48:53.734613Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-01T08:08:23.404839+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T08:08:23.404839+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14425","last_updated":"2025-06-05T00:49:05Z","snapshot_observed_at":"2026-07-06T20:39:43.403470Z","submitted_at":"2025-02-20T10:23:27Z","title":"A Survey on Data Contamination for Large Language Models","version":2},"cited_work":{"arxiv_id":"2502.14425","doi":"10.48550/arxiv.2502.14425","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14425","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2502.14425 , year=","venue":"ArXiv.org","work_id":"93c324c7-c726-4b33-a231-2ee7f4fb0fab","year":2025},"citing_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-12T13:48:53.691192Z"},"links":{"cited_paper":"/paper/2502.14425","citing_paper":"/paper/2509.16941"},"observation_digest":"sha256:2b1592e5acd7d96d9961f937ccf2a639cc6b6aa5330245ae8f05d1bfe1d094f0","observation_id":"86b4d980-55e6-415f-8f05-d037e9fda9a3","resolution":{"observed_at":"2026-05-12T13:48:53.741016Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.11425","last_updated":"2025-06-20T23:32:06Z","snapshot_observed_at":"2026-08-02T10:10:46.757463Z","submitted_at":"2025-06-13T02:46:53Z","title":"Agent-RLVR: Training Software Engineering Agents via Guidance and Environment Rewards","version":2},"cited_work":{"arxiv_id":"2506.11425","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.11425","snapshot_observed_at":"2026-07-03T20:18:55.557008Z","title":"Agent-rlvr: Training software engineering agents via guidance and environment rewards","venue":null,"work_id":"8004998c-b315-4791-bd76-2c438beeb63e","year":2025},"citing_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-12T13:48:53.691192Z"},"links":{"cited_paper":"/paper/2506.11425","citing_paper":"/paper/2509.16941"},"observation_digest":"sha256:58ded44a52bc7df469a7e7d87a78fd38352aa0256e7519d19305db1522348a33","observation_id":"781e070c-9bf1-4920-951e-e58502d4508d","resolution":{"observed_at":"2026-05-12T13:48:53.748215Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"44bb7c9f-32b1-4d3f-b9a6-b56e40511d42","year":2024},"citing_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-12T13:48:53.691192Z"},"links":{"citing_paper":"/paper/2509.16941"},"observation_digest":"sha256:a192207fa9cbeac5cd55cd9854213e2d24e724a3340ef229949219ece048159b","observation_id":"547ba755-0716-4334-9fd2-073fe81acbb7","resolution":{"observed_at":"2026-05-12T13:48:53.821287Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"f7d50b35-04da-4b0a-83df-112b4fbfe890","year":2009},"citing_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-12T13:48:53.691192Z"},"links":{"citing_paper":"/paper/2509.16941"},"observation_digest":"sha256:a49fb76dfe551019d74ba457e455e8ba2c4fa333a0983c1054a294f916d2dfc5","observation_id":"0f4ac406-f073-49d6-b2f8-711ac0c9b261","resolution":{"observed_at":"2026-05-12T13:48:53.824830Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Hendrycks, S","venue":null,"work_id":"cc3adf49-ac8c-4c17-aff0-ab3e8cb59070","year":2021},"citing_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-12T13:48:53.691192Z"},"links":{"citing_paper":"/paper/2509.16941"},"observation_digest":"sha256:a9574417e1d84c8d4f2e3c7992ac946fdcfd1bed9b6b3ec0fe6228d796ef18ca","observation_id":"bd7f8f2a-2500-45cf-bf98-c977351137b0","resolution":{"observed_at":"2026-05-12T13:48:53.829641Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"12aa18d5-0f9d-41df-b6c0-a9fe044485de","year":2024},"citing_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-12T13:48:53.691192Z"},"links":{"citing_paper":"/paper/2509.16941"},"observation_digest":"sha256:f54640623faf6364acf04694872bbb46caed00c0be94c46763b1beb0713721d2","observation_id":"61dfd1a7-6b83-4089-92fd-84cf07c1a199","resolution":{"observed_at":"2026-05-12T13:48:53.837044Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"URLhttps://openai.com/index/introducing-swe-bench-verified/","venue":null,"work_id":"9a79d230-d730-4247-ac9b-64d02cb7cb8a","year":2024},"citing_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-12T13:48:53.691192Z"},"links":{"citing_paper":"/paper/2509.16941"},"observation_digest":"sha256:9936ff4efde7be50b1719d5185b6721d415545b92eb2e93f012cdf91883c1b01","observation_id":"41215c8a-1442-4c82-8315-342ee5a72f8f","resolution":{"observed_at":"2026-05-12T13:48:53.840782Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Steidl, B","venue":null,"work_id":"682c0f99-8b98-4863-8d63-c27d62866d51","year":2017},"citing_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-12T13:48:53.691192Z"},"links":{"citing_paper":"/paper/2509.16941"},"observation_digest":"sha256:af0d4a0da8556166964a0e0c5129d27333bb43eff68e0d180fe641d7aa980990","observation_id":"a3f1750e-fb59-499c-b510-21b55d3c962b","resolution":{"observed_at":"2026-05-12T13:48:53.847696Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.19314","last_updated":"2025-04-18T19:36:00Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-27T16:47:42Z","title":"LiveBench: A Challenging, Contamination-Limited LLM Benchmark","version":2},"cited_work":{"arxiv_id":"2406.19314","doi":"10.48550/arxiv.2406.19314","metadata_source":"pith","pith_arxiv_id":"2406.19314","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LiveBench: A Challenging, Contamination-Limited LLM Benchmark","venue":"cs.CL","work_id":"6b2b33bf-350e-4ee2-b8a7-f011e53384e7","year":2024},"citing_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-12T13:48:53.691192Z"},"links":{"cited_paper":"/paper/2406.19314","citing_paper":"/paper/2509.16941"},"observation_digest":"sha256:409b1903b2ac46238c9dc254a2f3a60e6a6a320e2a5253530037c84be71fca13","observation_id":"c3a1428d-2403-40a5-b69c-8fc4fd377703","resolution":{"observed_at":"2026-05-15T04:48:26.603034Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.01489","last_updated":"2024-10-29T17:29:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-07-01T17:24:45Z","title":"Agentless: Demystifying LLM-based Software Engineering Agents","version":2},"cited_work":{"arxiv_id":"2407.01489","doi":"10.48550/arxiv.2407.01489","metadata_source":"pith","pith_arxiv_id":"2407.01489","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Agentless: Demystifying LLM-based Software Engineering Agents","venue":"cs.SE","work_id":"71c901c4-3c83-4e10-af54-3daef7fff397","year":2024},"citing_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-12T13:48:53.691192Z"},"links":{"cited_paper":"/paper/2407.01489","citing_paper":"/paper/2509.16941"},"observation_digest":"sha256:a255eed4a78d9acde65128b8fa3531817906f18a08d2896267876e5a16c2e71e","observation_id":"947d9844-7be4-42d4-9912-6e649fc11685","resolution":{"observed_at":"2026-05-12T13:48:53.760722Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:51.153796+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:51.153796+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04244","last_updated":"2024-06-06T16:41:39Z","snapshot_observed_at":"2026-07-30T15:43:06.151242Z","submitted_at":"2024-06-06T16:41:39Z","title":"Benchmark Data Contamination of Large Language Models: A Survey","version":1},"cited_work":{"arxiv_id":"2406.04244","doi":"10.48550/arxiv.2406.04244","metadata_source":"pith","pith_arxiv_id":"2406.04244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Benchmark Data Contamination of Large Language Models: A Survey","venue":"cs.CL","work_id":"30fe188d-51ed-4ae5-8557-dfd5c814931e","year":2024},"citing_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-12T13:48:53.691192Z"},"links":{"cited_paper":"/paper/2406.04244","citing_paper":"/paper/2509.16941"},"observation_digest":"sha256:cc053d4be7002700f1510f1e86ba510a199d8e2f5a44dd38e027329e731dd4ba","observation_id":"50e60ca1-3688-455e-a3a9-d82c29367e6e","resolution":{"observed_at":"2026-05-22T23:10:41.376209Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"e2eee463-f6de-496c-94fb-e9201d2e2979","year":2024},"citing_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-12T13:48:53.691192Z"},"links":{"citing_paper":"/paper/2509.16941"},"observation_digest":"sha256:5ffd45dd825383029fd6ac0a667106e7fc0f57b3f5784e152b91a7ba67c4d63c","observation_id":"eb4a0324-6647-4cc5-a03b-c60c208bf59d","resolution":{"observed_at":"2026-05-12T13:48:53.814455Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":"2410.03859","doi":"10.48550/arxiv.2410.03859","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv.org","venue":"arXiv (Cornell University)","work_id":"633853a8-4945-471e-8a0f-e3fd7fd02f77","year":2024},"citing_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-12T13:48:53.691192Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2509.16941"},"observation_digest":"sha256:36a5153b30bb320609e03cca0d33dc03e0fd31ddac9e0e2d503f701dbe010bcb","observation_id":"c816a82b-f825-466f-9d70-8f6afb83e9bc","resolution":{"observed_at":"2026-05-12T13:48:53.774713Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.02605","last_updated":"2024-11-11T09:45:11Z","snapshot_observed_at":"2026-08-05T03:59:55.350262Z","submitted_at":"2024-04-03T09:51:39Z","title":"A Gauss-Seidel method for solving multi-leader-multi-follower games","version":2},"cited_work":{"arxiv_id":"2404.02605","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.02605","snapshot_observed_at":"2026-07-03T08:57:48.222117Z","title":"arXiv preprint arXiv:2404.02605 , year=","venue":null,"work_id":"7ca3dbca-5170-4d97-b19b-fd065cfe01e0","year":2024},"citing_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-12T13:48:53.691192Z"},"links":{"cited_paper":"/paper/2404.02605","citing_paper":"/paper/2509.16941"},"observation_digest":"sha256:2e1f602618d113e5726fe00199d0f4671e54611c6f17b8f9a83244acd8415c88","observation_id":"3f0b4741-ac2d-4463-a347-5a80fe12f348","resolution":{"observed_at":"2026-05-12T13:48:53.781577Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23419","last_updated":"2025-06-01T14:53:40Z","snapshot_observed_at":"2026-08-02T15:14:08.960111Z","submitted_at":"2025-05-29T13:09:44Z","title":"SWE-bench Goes Live!","version":2},"cited_work":{"arxiv_id":"2505.23419","doi":"10.48550/arxiv.2505.23419","metadata_source":"pith","pith_arxiv_id":"2505.23419","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Swe-bench goes live!","venue":"cs.SE","work_id":"648e52f0-7b78-4b75-8b1d-e89c9ba63afe","year":2025},"citing_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-12T13:48:53.691192Z"},"links":{"cited_paper":"/paper/2505.23419","citing_paper":"/paper/2509.16941"},"observation_digest":"sha256:64f1ccc2d90eb270bf30fba2a73afddc1cc93e32d58ed9cb98a874c1e5de8688","observation_id":"a9f94058-bd67-404e-a148-0ea32a88f3b1","resolution":{"observed_at":"2026-05-12T13:48:53.788715Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00332","last_updated":"2024-11-22T22:27:49Z","snapshot_observed_at":"2026-07-06T18:08:04.815730Z","submitted_at":"2024-05-01T05:52:05Z","title":"A Careful Examination of Large Language Model Performance on Grade School Arithmetic","version":4},"cited_work":{"arxiv_id":"2405.00332","doi":"10.48550/arxiv.2405.00332","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00332","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2405.00332 , year=","venue":"arXiv (Cornell University)","work_id":"172a8232-f346-4451-ab9a-b76d33c6c118","year":2024},"citing_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-12T13:48:53.691192Z"},"links":{"cited_paper":"/paper/2405.00332","citing_paper":"/paper/2509.16941"},"observation_digest":"sha256:239937c888b8fb759618a56403b5cfb9dd6e0ce64f9b0a38ee336827a6408b07","observation_id":"69e784c3-d604-4142-8de4-1b074e2caeaa","resolution":{"observed_at":"2026-05-12T13:48:53.795243Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Book 978","venue":null,"work_id":"562ba7e7-675e-45de-baa8-8f1a47586f65","year":2024},"citing_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-12T13:48:53.691192Z"},"links":{"citing_paper":"/paper/2509.16941"},"observation_digest":"sha256:5117fd5af50d7f36d7d0ac65cfd9cdf4e7167fc9cf816eda2dae1602b205f617","observation_id":"a8e8c8fb-7d89-4c65-993b-71d55550c1a1","resolution":{"observed_at":"2026-05-12T13:48:53.818015Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"iOS Contacts)","venue":null,"work_id":"9dc6d0ba-87a4-49d2-8414-0cb2fd4cf926","year":null},"citing_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-12T13:48:53.691192Z"},"links":{"citing_paper":"/paper/2509.16941"},"observation_digest":"sha256:ec524eabc6a20fdb3efe1140d6037681414194a6f804026c577269fb98176b2c","observation_id":"0f712ad7-fe0a-4caa-a996-1993c4edd292","resolution":{"observed_at":"2026-05-12T13:48:53.799271Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"764a827b-b6ea-42c5-a56d-6bdcf2739cf6","year":null},"citing_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-12T13:48:53.691192Z"},"links":{"citing_paper":"/paper/2509.16941"},"observation_digest":"sha256:7c5485fcea377788edbc691176077d4629eff2f1bc1e28501f58e8ab143c155f","observation_id":"e24d14d4-c7ef-4ac7-97fb-0d7918ec8ef3","resolution":{"observed_at":"2026-05-12T13:48:53.802767Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"No actions recorded","venue":null,"work_id":"6bf5248c-198b-413e-8750-392981f5da38","year":null},"citing_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-12T13:48:53.691192Z"},"links":{"citing_paper":"/paper/2509.16941"},"observation_digest":"sha256:26f4cc44f6dc70aa247d12e7659918c0092129eaa456289ab4cf3083b37ba4a8","observation_id":"47a54694-2997-4dba-94c4-cf6eaad0c878","resolution":{"observed_at":"2026-05-12T13:48:53.810753Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","latest_version":2,"primary_category":"cs.SE","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?"},"reference_resolution":{"displayed":24,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":5,"verified_exact":12,"verified_fuzzy":7},"total_outbound_references":24},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 24 of 24 outbound references and 100 inbound Pith citation observations for arXiv:2509.16941."}