{"as_of":"2026-08-07T10:11:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:c46adbd11bc1207a16fe7622ba89e44ca03f2e5380691f000a14f2160cf0f105","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":40,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":40,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":40,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":40,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T06:00:17.468272Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":1,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":"2410.03859","doi":"10.48550/arxiv.2410.03859","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv.org","venue":"arXiv (Cornell University)","work_id":"633853a8-4945-471e-8a0f-e3fd7fd02f77","year":2024},"citing_paper":{"arxiv_id":"2502.10517","last_updated":"2025-02-14T19:30:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-14T19:30:53Z","title":"KernelBench: Can LLMs Write Efficient GPU Kernels?","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-15T16:55:01.976356Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2502.10517"},"observation_digest":"sha256:be351f9df6304e3150db7ee6305d8c4e5faec8718abd5823c1464eb5d8b6529b","observation_id":"569e9854-eada-48dd-817c-1d3e64809d0a","resolution":{"observed_at":"2026-05-15T16:55:02.102197Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":"2410.03859","doi":"10.48550/arxiv.2410.03859","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv.org","venue":"arXiv (Cornell University)","work_id":"633853a8-4945-471e-8a0f-e3fd7fd02f77","year":2024},"citing_paper":{"arxiv_id":"2502.18449","last_updated":"2025-12-01T00:16:59Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-25T18:45:04Z","title":"SWE-RL: Advancing LLM Reasoning via Reinforcement Learning on Open Software Evolution","version":2},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-05-15T10:27:56.185943Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2502.18449"},"observation_digest":"sha256:3ac754e5621142dea8ea01c52cd4ffa8a84c701c43fd371ce5f720192cc90a18","observation_id":"0182c42b-11f8-4bfa-9994-6139a69d8496","resolution":{"observed_at":"2026-05-15T10:27:56.346568Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":"2410.03859","doi":"10.48550/arxiv.2410.03859","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv.org","venue":"arXiv (Cornell University)","work_id":"633853a8-4945-471e-8a0f-e3fd7fd02f77","year":2024},"citing_paper":{"arxiv_id":"2506.02387","last_updated":"2026-04-13T08:26:12Z","snapshot_observed_at":"2026-08-04T07:16:14.991803Z","submitted_at":"2025-06-03T02:57:38Z","title":"VS-Bench: Evaluating VLMs for Strategic Abilities in Multi-Agent Environments","version":3},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-05-19T11:57:08.314088Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2506.02387"},"observation_digest":"sha256:b57566e3f371568b6266af759cfc3bc73435a0c763f13c143f8d21d25d9b2688","observation_id":"bb7b2022-6b02-4d15-b491-9a50cc08c345","resolution":{"observed_at":"2026-05-19T11:57:16.238337Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-07T05:26:15.130136Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.07964","last_updated":"2025-06-09T17:39:48Z","snapshot_observed_at":"2026-08-07T05:19:06.277008Z","submitted_at":"2025-06-09T17:39:48Z","title":"SlideCoder: Layout-aware RAG-enhanced Hierarchical Slide Generation from Design","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-07T05:26:15.130136Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2506.07964"},"observation_digest":"sha256:1bd731ae94ec6fc6c6ecd80385e918b4d4676b5148dcd0a74d2f01701ecb2e49","observation_id":"c65b9d81-33ba-4eb1-ae19-eae98d3585b1","resolution":{"observed_at":"2026-08-07T05:26:15.130136Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-07T06:00:17.468272Z","title":"Swe-bench multimodal: Do ai systems generalize to visual software domains?","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.11102","last_updated":"2025-06-06T17:52:18Z","snapshot_observed_at":"2026-08-07T05:54:56.593167Z","submitted_at":"2025-06-06T17:52:18Z","title":"Evolutionary Perspectives on the Evaluation of LLM-Based AI Agents: A Comprehensive Survey","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T06:00:17.468272Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2506.11102"},"observation_digest":"sha256:58001c05a2616106d9cb4f38462b0296c4266adb1b18803c77f4fa3f23d12ad6","observation_id":"9a21d675-1c13-4335-81f9-84ac9f795873","resolution":{"observed_at":"2026-08-07T06:00:17.468272Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-06T18:34:31.618617Z","title":"Swe-bench multimodal: Do ai systems generalize to visual software domains?arXiv preprint arXiv:2410.03859, 2024","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.08149","last_updated":"2025-09-13T14:59:53Z","snapshot_observed_at":"2026-08-07T09:24:11.181871Z","submitted_at":"2025-07-10T20:12:54Z","title":"Code with Me or for Me? How Increasing AI Automation Transforms Developer Workflows","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T18:34:31.618617Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2507.08149"},"observation_digest":"sha256:8de17c86409756f3acfaa8fa6f39696e3614859ea3a85a71d13857b82fc70299","observation_id":"fe719371-92cc-4834-a8a2-9e075993215e","resolution":{"observed_at":"2026-08-06T18:34:31.618617Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-06T18:19:58.357081Z","title":"Swe-bench multimodal: Do ai systems generalize to visual software domains? arXiv preprint arXiv:2410.03859, 2024","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.08719","last_updated":"2025-07-11T16:19:53Z","snapshot_observed_at":"2026-08-07T04:09:15.037517Z","submitted_at":"2025-07-11T16:19:53Z","title":"Multilingual Multimodal Software Developer for Code Generation","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-06T18:19:58.357081Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2507.08719"},"observation_digest":"sha256:1b7d3d4680c744e68a15ecd90f303db8dcdfb450ed2df199c85ebe80db9a8930","observation_id":"9ae7f4e5-42dd-471d-b023-015998974436","resolution":{"observed_at":"2026-08-06T18:19:58.357081Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":"2410.03859","doi":"10.48550/arxiv.2410.03859","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv.org","venue":"arXiv (Cornell University)","work_id":"633853a8-4945-471e-8a0f-e3fd7fd02f77","year":2024},"citing_paper":{"arxiv_id":"2507.15003","last_updated":"2025-07-20T15:15:58Z","snapshot_observed_at":"2026-07-06T21:59:59.698990Z","submitted_at":"2025-07-20T15:15:58Z","title":"The Rise of AI Teammates in Software Engineering (SE) 3.0: How Autonomous Coding Agents Are Reshaping Software Engineering","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-15T17:38:55.159673Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2507.15003"},"observation_digest":"sha256:8e279f1683ea95977ebb40f2aa1bcbb74f740c32819c18a25abb5c6bc326f477","observation_id":"3789f81f-780c-4954-b3d2-29b734b7aab6","resolution":{"observed_at":"2026-05-15T17:38:55.375974Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":"2410.03859","doi":"10.48550/arxiv.2410.03859","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv.org","venue":"arXiv (Cornell University)","work_id":"633853a8-4945-471e-8a0f-e3fd7fd02f77","year":2024},"citing_paper":{"arxiv_id":"2507.21046","last_updated":"2026-01-16T20:59:08Z","snapshot_observed_at":"2026-08-01T06:32:44.461162Z","submitted_at":"2025-07-28T17:59:05Z","title":"A Survey of Self-Evolving Agents: What, When, How, and Where to Evolve on the Path to Artificial Super Intelligence","version":4},"reference_index":109,"source":"arxiv_source","source_observed_at":"2026-05-14T22:23:14.621091Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2507.21046"},"observation_digest":"sha256:f1c69825a6399da1bda28397852fd64bd0c306269bf7fe160cfb5a6c0a39acf4","observation_id":"0521b7dd-b328-4137-8a6e-81d155a0e877","resolution":{"observed_at":"2026-05-14T22:23:15.799098Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T10:29:20.689824Z","title":"Jimenez, Alex L","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.05372","last_updated":"2026-01-26T08:01:41Z","snapshot_observed_at":"2026-08-05T23:14:14.618484Z","submitted_at":"2025-09-04T09:41:57Z","title":"Adversarial Bug Reports as a Security Risk in Language Model-Based Automated Program Repair","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-05T10:29:20.689824Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2509.05372"},"observation_digest":"sha256:3898221367008d285444f2c8b7d321ec8f3f1b57d9eeb5cedfdf4928d226c5f8","observation_id":"ad34d2a1-8cf0-47ca-92de-de061c9c6bc5","resolution":{"observed_at":"2026-08-05T10:29:20.689824Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":"2410.03859","doi":"10.48550/arxiv.2410.03859","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv.org","venue":"arXiv (Cornell University)","work_id":"633853a8-4945-471e-8a0f-e3fd7fd02f77","year":2024},"citing_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-06T21:34:46.041153Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-12T13:48:53.691192Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2509.16941"},"observation_digest":"sha256:bfbb3330af65ad60073179783ebc9df915a931691af34d4312462068cf938101","observation_id":"c816a82b-f825-466f-9d70-8f6afb83e9bc","resolution":{"observed_at":"2026-05-12T13:48:53.774713Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-04T10:36:40.623619Z","title":"Swe-bench multimodal: Do ai systems generalize to visual software domains?arXiv preprint arXiv:2410.03859, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2510.09801","last_updated":"2026-06-09T14:05:10Z","snapshot_observed_at":"2026-08-07T09:24:40.875565Z","submitted_at":"2025-10-10T19:04:28Z","title":"How can we assess human-agent interactions? Case studies in software agent design","version":3},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-04T10:36:40.623619Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2510.09801"},"observation_digest":"sha256:2de643a350d74497a979353dfe58e5cbd05a3a56809258ef72b734eea4d23428","observation_id":"95d2e266-5e50-487d-bc5f-4125e7126727","resolution":{"observed_at":"2026-08-04T10:36:40.623619Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":"2410.03859","doi":"10.48550/arxiv.2410.03859","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv.org","venue":"arXiv (Cornell University)","work_id":"633853a8-4945-471e-8a0f-e3fd7fd02f77","year":2024},"citing_paper":{"arxiv_id":"2512.18470","last_updated":"2026-04-04T09:52:04Z","snapshot_observed_at":"2026-07-30T13:30:43.108201Z","submitted_at":"2025-12-20T19:08:15Z","title":"SWE-EVO: Benchmarking Coding Agents in Long-Horizon Software Evolution Scenarios","version":5},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-05-16T20:24:40.939455Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2512.18470"},"observation_digest":"sha256:bb295e92880d8d392218b5f3a36ad7f7725ca50a20111d76ee4c261b572aa677","observation_id":"1039a3cb-282b-4028-9ccb-226845f49d14","resolution":{"observed_at":"2026-05-16T20:28:24.528054Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":"2410.03859","doi":"10.48550/arxiv.2410.03859","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv.org","venue":"arXiv (Cornell University)","work_id":"633853a8-4945-471e-8a0f-e3fd7fd02f77","year":2024},"citing_paper":{"arxiv_id":"2601.11848","last_updated":"2026-04-06T02:33:13Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T00:34:32Z","title":"Compass vs Railway Tracks: Unpacking User Mental Models for Communicating Long-Horizon Work to Humans vs. AI","version":2},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-05-16T14:09:33.786576Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2601.11848"},"observation_digest":"sha256:8f013ad0f0b216a05139a32acf25eca625d467bd6bb4d053d4fa37d417e39811","observation_id":"66cdd8a2-8d5a-4c27-bc17-cbdd02e06549","resolution":{"observed_at":"2026-05-16T14:11:01.384484Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-02T23:31:17.421699Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.13723","last_updated":"2026-07-09T17:16:26Z","snapshot_observed_at":"2026-08-07T00:34:02.672460Z","submitted_at":"2026-02-14T11:07:58Z","title":"Compiling Large Multi-Modal Requirement Documents into Runnable Software Systems: From an Agentic Test-Driven Perspective","version":5},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-02T23:31:17.421699Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2602.13723"},"observation_digest":"sha256:c796afa3f8ef3b5ddff2cb1945cbb3026ef16fecb93b5dd2e17e620c070aa6f5","observation_id":"d3ebc9cc-6ce8-49bf-a4b2-957fe8b7779d","resolution":{"observed_at":"2026-08-02T23:31:17.421699Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":"2410.03859","doi":"10.48550/arxiv.2410.03859","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv.org","venue":"arXiv (Cornell University)","work_id":"633853a8-4945-471e-8a0f-e3fd7fd02f77","year":2024},"citing_paper":{"arxiv_id":"2604.12162","last_updated":"2026-04-14T00:43:20Z","snapshot_observed_at":"2026-07-31T20:58:59.645474Z","submitted_at":"2026-04-14T00:43:20Z","title":"AlphaEval: Evaluating Agents in Production","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-10T16:30:51.886471Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2604.12162"},"observation_digest":"sha256:b39154a9d531f6e39fa7a2b0d06df8fabd0514b1511ad5544caea3454bdd187a","observation_id":"6e31f8ef-6b5a-44c4-8f5d-6d244db49fe9","resolution":{"observed_at":"2026-05-11T08:45:58.877213Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":"2410.03859","doi":"10.48550/arxiv.2410.03859","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv.org","venue":"arXiv (Cornell University)","work_id":"633853a8-4945-471e-8a0f-e3fd7fd02f77","year":2024},"citing_paper":{"arxiv_id":"2604.26275","last_updated":"2026-04-29T04:06:47Z","snapshot_observed_at":"2026-07-06T23:11:57.238808Z","submitted_at":"2026-04-29T04:06:47Z","title":"Agentic AI in the Software Development Lifecycle: Architecture, Empirical Evidence, and the Reshaping of Software Engineering","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-07T13:26:26.648410Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2604.26275"},"observation_digest":"sha256:f0c836ebb665ffcd822aa61a6f6788db993fe095f4fcde4712b75f73937d4ea1","observation_id":"7d13fd6b-bae3-476c-9a8b-788fc466c5fd","resolution":{"observed_at":"2026-05-12T08:56:26.738521Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":"2410.03859","doi":"10.48550/arxiv.2410.03859","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv.org","venue":"arXiv (Cornell University)","work_id":"633853a8-4945-471e-8a0f-e3fd7fd02f77","year":2024},"citing_paper":{"arxiv_id":"2605.02244","last_updated":"2026-05-04T05:37:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-05-04T05:37:36Z","title":"The Conversations Beneath the Code: Triadic Data for Long-Horizon Software Engineering Agents","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-05-08T18:30:21.856024Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2605.02244"},"observation_digest":"sha256:b2fe797b27f31d1333acc5addebf59ae5e8a2d364b9ef8ea942f554e613a4a25","observation_id":"bd71bb2e-2a45-426d-a00d-50930f3ba827","resolution":{"observed_at":"2026-05-09T06:25:39.708557Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":"2410.03859","doi":"10.48550/arxiv.2410.03859","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv.org","venue":"arXiv (Cornell University)","work_id":"633853a8-4945-471e-8a0f-e3fd7fd02f77","year":2024},"citing_paper":{"arxiv_id":"2605.09423","last_updated":"2026-05-13T08:34:40Z","snapshot_observed_at":"2026-07-31T05:53:49.791813Z","submitted_at":"2026-05-10T08:51:50Z","title":"SimWorld Studio: Automatic Environment Generation with Evolving Coding Agent for Embodied Agent Learning","version":1},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-05-12T04:21:44.087943Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2605.09423"},"observation_digest":"sha256:6bce883adf48ffc779a2652de9d3e6d48545d75726ab9b7e9a6a7d28c965693a","observation_id":"5356b90d-e7b9-4a66-a319-e9d47b33bc26","resolution":{"observed_at":"2026-05-12T06:21:26.539858Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":"2410.03859","doi":"10.48550/arxiv.2410.03859","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv.org","venue":"arXiv (Cornell University)","work_id":"633853a8-4945-471e-8a0f-e3fd7fd02f77","year":2024},"citing_paper":{"arxiv_id":"2605.09423","last_updated":"2026-05-13T08:34:40Z","snapshot_observed_at":"2026-07-31T05:53:49.791813Z","submitted_at":"2026-05-10T08:51:50Z","title":"SimWorld Studio: Automatic Environment Generation with Evolving Coding Agent for Embodied Agent Learning","version":2},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-05-14T21:30:42.766390Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2605.09423"},"observation_digest":"sha256:884c92f2f9ea8eba3339d7e865f3e9be3afd768ae86fb68768054b85a96eb529","observation_id":"085178c1-03f3-4b6f-956a-f03900be35ec","resolution":{"observed_at":"2026-05-14T21:32:59.504330Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":"2410.03859","doi":"10.48550/arxiv.2410.03859","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv.org","venue":"arXiv (Cornell University)","work_id":"633853a8-4945-471e-8a0f-e3fd7fd02f77","year":2024},"citing_paper":{"arxiv_id":"2605.14415","last_updated":"2026-05-14T06:04:40Z","snapshot_observed_at":"2026-07-06T23:25:49.545418Z","submitted_at":"2026-05-14T06:04:40Z","title":"SWE-Chain: Benchmarking Coding Agents on Chained Release-Level Package Upgrades","version":1},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-05-15T02:31:18.183715Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2605.14415"},"observation_digest":"sha256:227cf63c702a99a58788939186438f14883201ea7ba22b1ef21e29857a50d949","observation_id":"7a779983-db03-424a-b2f2-10e14520202f","resolution":{"observed_at":"2026-05-15T02:33:32.348124Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":"2410.03859","doi":"10.48550/arxiv.2410.03859","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv.org","venue":"arXiv (Cornell University)","work_id":"633853a8-4945-471e-8a0f-e3fd7fd02f77","year":2024},"citing_paper":{"arxiv_id":"2605.20520","last_updated":"2026-05-19T21:42:32Z","snapshot_observed_at":"2026-08-03T02:30:24.579386Z","submitted_at":"2026-05-19T21:42:32Z","title":"Open-World Evaluations for Measuring Frontier AI Capabilities","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-21T06:38:51.427985Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2605.20520"},"observation_digest":"sha256:02f9aa3d63eb856e3eda5e6e55955e30fc5341abb1b489242776c4eada16e872","observation_id":"41f1a7c4-cf12-47f7-ad24-997f0c3ee081","resolution":{"observed_at":"2026-05-21T06:39:43.750853Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":"2410.03859","doi":"10.48550/arxiv.2410.03859","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv.org","venue":"arXiv (Cornell University)","work_id":"633853a8-4945-471e-8a0f-e3fd7fd02f77","year":2024},"citing_paper":{"arxiv_id":"2605.30690","last_updated":"2026-05-29T00:34:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-05-29T00:34:40Z","title":"ElasticMem: Latent Memory as a Learnable Resource for LLM Agents","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-06-28T23:06:57.377183Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2605.30690"},"observation_digest":"sha256:75edd23624aec42151202f678f9cf6bf4bf1577ea1e586ea194c87939ca94aea","observation_id":"b5bdc529-81b9-402c-8587-8a1d835b3434","resolution":{"observed_at":"2026-06-29T00:22:51.863309Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":"2410.03859","doi":"10.48550/arxiv.2410.03859","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv.org","venue":"arXiv (Cornell University)","work_id":"633853a8-4945-471e-8a0f-e3fd7fd02f77","year":2024},"citing_paper":{"arxiv_id":"2606.00750","last_updated":"2026-05-30T14:34:02Z","snapshot_observed_at":"2026-07-06T23:41:24.911421Z","submitted_at":"2026-05-30T14:34:02Z","title":"I-WebGenBench : Evaluating Interactivity in LLM-Generated Scientific Web Applications","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-06-28T18:53:18.645984Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2606.00750"},"observation_digest":"sha256:708908437d6fdedf13474a972fcb6eb83607673b6485cfa4361072fe5afb88cd","observation_id":"ccc90743-de5f-4748-b3b5-bdb7d7df450e","resolution":{"observed_at":"2026-06-28T19:42:36.256892Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":"2410.03859","doi":"10.48550/arxiv.2410.03859","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv.org","venue":"arXiv (Cornell University)","work_id":"633853a8-4945-471e-8a0f-e3fd7fd02f77","year":2024},"citing_paper":{"arxiv_id":"2606.00832","last_updated":"2026-05-30T18:08:51Z","snapshot_observed_at":"2026-07-06T23:41:30.158879Z","submitted_at":"2026-05-30T18:08:51Z","title":"Momento: Evaluating Persistent Memory and Reasoning with Multi-Session Agentic Conversations","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-28T18:44:12.893087Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2606.00832"},"observation_digest":"sha256:b76ecd01e6519646b42c03322575ea70f4ade8cc23ef361a7befbed95d6601a0","observation_id":"33d5f3d5-9b40-45c6-b172-dab231b14e54","resolution":{"observed_at":"2026-06-28T20:22:37.804536Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":"2410.03859","doi":"10.48550/arxiv.2410.03859","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv.org","venue":"arXiv (Cornell University)","work_id":"633853a8-4945-471e-8a0f-e3fd7fd02f77","year":2024},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:cd5887e54083343b38ed52baf312dd531427cce2488e6e68bb97cdd107ca2d6d","observation_id":"bb2696f8-df19-4e7c-90af-a0223aacb4d4","resolution":{"observed_at":"2026-07-02T03:36:29.215914Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":"2410.03859","doi":"10.48550/arxiv.2410.03859","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv.org","venue":"arXiv (Cornell University)","work_id":"633853a8-4945-471e-8a0f-e3fd7fd02f77","year":2024},"citing_paper":{"arxiv_id":"2606.05920","last_updated":"2026-06-04T09:24:30Z","snapshot_observed_at":"2026-08-06T16:42:33.524768Z","submitted_at":"2026-06-04T09:24:30Z","title":"Asuka-Bench: Benchmarking Code Agents on Underspecified User Intent and Multi-Round Refinement","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-06-28T00:26:22.041924Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2606.05920"},"observation_digest":"sha256:7942febaec226785c62c4981716c4a224ba8014bbaefabca7860101173196ef8","observation_id":"38fa32d3-1c5a-4cff-b2b0-c8185b33f454","resolution":{"observed_at":"2026-07-02T14:27:04.283039Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":"2410.03859","doi":"10.48550/arxiv.2410.03859","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv.org","venue":"arXiv (Cornell University)","work_id":"633853a8-4945-471e-8a0f-e3fd7fd02f77","year":2024},"citing_paper":{"arxiv_id":"2606.10106","last_updated":"2026-06-08T19:35:37Z","snapshot_observed_at":"2026-08-01T08:26:23.637148Z","submitted_at":"2026-06-08T19:35:37Z","title":"What makes a harness a harness: necessary and sufficient conditions for an agent harness","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-06-27T15:15:57.372858Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2606.10106"},"observation_digest":"sha256:11d5bbdaa77e06f94fd51eccf5a7c907feff84b77e24e4bd13038f7374dd1120","observation_id":"19d84601-11dd-49dc-aa1c-5df8de9866df","resolution":{"observed_at":"2026-06-27T15:21:00.801996Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":"2410.03859","doi":"10.48550/arxiv.2410.03859","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv.org","venue":"arXiv (Cornell University)","work_id":"633853a8-4945-471e-8a0f-e3fd7fd02f77","year":2024},"citing_paper":{"arxiv_id":"2606.11470","last_updated":"2026-06-09T21:59:37Z","snapshot_observed_at":"2026-07-06T23:50:35.052764Z","submitted_at":"2026-06-09T21:59:37Z","title":"The Periodic Table of LLM Reasoning: A Structured Survey of Reasoning Paradigms, Methods, and Failure Modes","version":1},"reference_index":282,"source":"arxiv_source","source_observed_at":"2026-06-27T12:59:51.091008Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2606.11470"},"observation_digest":"sha256:e4e8ccff34c063b6f76cb8881f7244b57b55223ff6c97376aecca22142f2e287","observation_id":"b70e4fc8-eb15-4695-8bf6-f249c480ea16","resolution":{"observed_at":"2026-07-03T05:57:41.639775Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":"2410.03859","doi":"10.48550/arxiv.2410.03859","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv.org","venue":"arXiv (Cornell University)","work_id":"633853a8-4945-471e-8a0f-e3fd7fd02f77","year":2024},"citing_paper":{"arxiv_id":"2606.12191","last_updated":"2026-06-10T15:15:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-06-10T15:15:01Z","title":"Agentic Environment Engineering for Large Language Models: A Survey of Environment Modeling, Synthesis, Evaluation, and Application","version":1},"reference_index":146,"source":"pdf_text","source_observed_at":"2026-06-27T09:46:30.702256Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2606.12191"},"observation_digest":"sha256:6d18c5bfb4ead2664328dd63ed1f74cd82d1130e55ffee9a62a9a4dfd07f4fa2","observation_id":"8d02e1bb-f5eb-4d85-a741-0e422d144021","resolution":{"observed_at":"2026-07-03T10:58:02.881675Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":"2410.03859","doi":"10.48550/arxiv.2410.03859","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv.org","venue":"arXiv (Cornell University)","work_id":"633853a8-4945-471e-8a0f-e3fd7fd02f77","year":2024},"citing_paper":{"arxiv_id":"2606.17454","last_updated":"2026-06-17T04:51:06Z","snapshot_observed_at":"2026-08-02T05:33:09.695431Z","submitted_at":"2026-06-16T03:17:03Z","title":"Dissecting model behavior through agent trajectories","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-06-27T01:27:39.812496Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2606.17454"},"observation_digest":"sha256:df5d570ac0ce8685ccfeaa80739043c050ea87b5256e3e7416a73b37ebd218e4","observation_id":"3c8df488-9523-4581-9792-dc4616f0332a","resolution":{"observed_at":"2026-07-03T20:18:56.805422Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":"2410.03859","doi":"10.48550/arxiv.2410.03859","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv.org","venue":"arXiv (Cornell University)","work_id":"633853a8-4945-471e-8a0f-e3fd7fd02f77","year":2024},"citing_paper":{"arxiv_id":"2606.17799","last_updated":"2026-07-18T22:14:13Z","snapshot_observed_at":"2026-08-02T11:05:59.500924Z","submitted_at":"2026-06-16T11:21:01Z","title":"Position: Coding Benchmarks Are Misaligned with Agentic Software Engineering","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-06-26T23:48:58.497927Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2606.17799"},"observation_digest":"sha256:909e413b9aca5802f111ebe6718052633654579f68654de282bdba9511879667","observation_id":"8a038349-a69d-4a13-9a3b-b18c1670b277","resolution":{"observed_at":"2026-07-03T22:08:58.945492Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-02T11:06:03.963144Z","title":"Jimenez, Alex L","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.17799","last_updated":"2026-07-18T22:14:13Z","snapshot_observed_at":"2026-08-02T11:05:59.500924Z","submitted_at":"2026-06-16T11:21:01Z","title":"Position: Coding Benchmarks Are Misaligned with Agentic Software Engineering","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-02T11:06:03.963144Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2606.17799"},"observation_digest":"sha256:120cc627e3fbd4e96ba6271304b817f048b11cee1e098a0ebeef5063c424bf2a","observation_id":"2f80db8c-a37a-4106-98e8-666af8c28a14","resolution":{"observed_at":"2026-08-02T11:06:03.963144Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":"2410.03859","doi":"10.48550/arxiv.2410.03859","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv.org","venue":"arXiv (Cornell University)","work_id":"633853a8-4945-471e-8a0f-e3fd7fd02f77","year":2024},"citing_paper":{"arxiv_id":"2606.18284","last_updated":"2026-06-10T02:04:29Z","snapshot_observed_at":"2026-07-06T23:53:45.117607Z","submitted_at":"2026-06-10T02:04:29Z","title":"Breaking the Solver Bottleneck: Training Task Generators at the Learnable Frontier","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-06-27T10:36:09.211639Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2606.18284"},"observation_digest":"sha256:f2f1793f6022da4e9194ec59241154c2b0033b5a3ccc4d73fda491461220539f","observation_id":"41139610-86fb-4ddb-b748-6ef934e23745","resolution":{"observed_at":"2026-07-03T08:57:48.236521Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":"2410.03859","doi":"10.48550/arxiv.2410.03859","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv.org","venue":"arXiv (Cornell University)","work_id":"633853a8-4945-471e-8a0f-e3fd7fd02f77","year":2024},"citing_paper":{"arxiv_id":"2606.19613","last_updated":"2026-06-17T21:36:09Z","snapshot_observed_at":"2026-08-06T20:08:22.959123Z","submitted_at":"2026-06-17T21:36:09Z","title":"StaminaBench: Stress-Testing Coding Agents over 100 Interaction Turns","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-06-26T19:47:59.090249Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2606.19613"},"observation_digest":"sha256:a36a9c4fa1e53f0d98dc4596e151a778b95d88835c3c376bd9e9bba4c0d3bc07","observation_id":"e12773f3-bcc7-4168-9eae-0b672d7bcb66","resolution":{"observed_at":"2026-07-04T02:19:23.939382Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":"2410.03859","doi":"10.48550/arxiv.2410.03859","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv.org","venue":"arXiv (Cornell University)","work_id":"633853a8-4945-471e-8a0f-e3fd7fd02f77","year":2024},"citing_paper":{"arxiv_id":"2606.25514","last_updated":"2026-06-24T07:48:05Z","snapshot_observed_at":"2026-08-03T07:55:38.164622Z","submitted_at":"2026-06-24T07:48:05Z","title":"Unlocking Model Potentials Through Adaptive Multi-Agent Scaffolding for Efficient Issue Resolution","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-06-25T20:15:00.517571Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2606.25514"},"observation_digest":"sha256:12db7bc6f67089129705194a5cc00bc91fda98264d98cd628fa59ce830a29850","observation_id":"59593680-b4e8-4ac0-adfe-660c1024b766","resolution":{"observed_at":"2026-07-04T20:20:07.769919Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":"2410.03859","doi":"10.48550/arxiv.2410.03859","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv.org","venue":"arXiv (Cornell University)","work_id":"633853a8-4945-471e-8a0f-e3fd7fd02f77","year":2024},"citing_paper":{"arxiv_id":"2607.00053","last_updated":"2026-06-30T01:46:26Z","snapshot_observed_at":"2026-07-07T00:05:41.920778Z","submitted_at":"2026-06-30T01:46:26Z","title":"SWE-Router: Routing in Multi-turn Agentic Software Engineering Tasks","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-07-02T18:19:43.146102Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2607.00053"},"observation_digest":"sha256:b243a53a51d4ee803e9d8e4aa93a6f8403ca9ee64175d0ce7b3cef81c8d1aa0a","observation_id":"6c9d3edf-dcc3-4f85-8f1b-787e72c9847a","resolution":{"observed_at":"2026-07-02T18:37:16.506209Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-04T00:55:46.348120Z","title":"arXiv preprint arXiv:2410.03859 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00267","last_updated":"2026-07-31T20:18:25Z","snapshot_observed_at":"2026-08-06T23:12:17.662506Z","submitted_at":"2026-07-31T20:18:25Z","title":"LoopsBench: From Harness Engineering to Loop Engineering in Benchmarking Coding Agent","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-04T00:55:46.348120Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2608.00267"},"observation_digest":"sha256:d17567757f11f45e112270023632e56fba10613128aa84e1eedd990e5330e090","observation_id":"f4605d47-9cd8-480f-9cc4-c8952400f1ea","resolution":{"observed_at":"2026-08-04T00:55:46.348120Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T18:29:10.457483Z","title":"E.; Zhang, A","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.03474","last_updated":"2026-08-04T11:14:00Z","snapshot_observed_at":"2026-08-07T09:13:09.209992Z","submitted_at":"2026-08-04T11:14:00Z","title":"MT-Web2Code: Benchmarking Coding Agents on Multi-Turn Regional Reconstruction and Localized Modification","version":1},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-05T18:29:10.457483Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2608.03474"},"observation_digest":"sha256:927176d550193c48ee62f61275acdd817e3d145fd7362af813df21900c2bcc96","observation_id":"c428be79-3082-463a-a539-fe217ca6d6cf","resolution":{"observed_at":"2026-08-05T18:29:10.457483Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-06T19:02:40.658366Z","title":"Swe-bench multimodal: Do ai systems generalize to visual software domains? arXiv preprint arXiv:2410.03859, 2024 b","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04682","last_updated":"2026-08-05T10:50:01Z","snapshot_observed_at":"2026-08-07T09:17:36.587461Z","submitted_at":"2026-08-05T10:50:01Z","title":"Active-SWE: Benchmarking Coding Agents for Proactive Bug Fixing without Issue Reports","version":1},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-08-06T19:02:40.658366Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2608.04682"},"observation_digest":"sha256:88b5e4d2f80171a099b3cb348eca8da707383ebeee59d62ea7825f7a2f9e5f9f","observation_id":"474521f1-6871-4316-9674-9fed3c956337","resolution":{"observed_at":"2026-08-06T19:02:40.658366Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2410.03859/citation-record","integrity":"/paper/2410.03859/integrity","json":"/paper/2410.03859/citation-record.json","paper":"/paper/2410.03859"},"outbound":[],"paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 40 inbound Pith citation observations for arXiv:2410.03859."}