{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZRQNUQ3IYWH5IVP4JUSW4MXMY7","short_pith_number":"pith:ZRQNUQ3I","schema_version":"1.0","canonical_sha256":"cc60da4368c58fd455fc4d256e32ecc7fe30c3e0af860554d16b9212ea2b3b63","source":{"kind":"arxiv","id":"2505.23419","version":2},"attestation_state":"computed","paper":{"title":"SWE-bench Goes Live!","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.SE","authors_text":"Bowen Li, Chaoyun Zhang, Chengxing Xie, Dongmei Zhang, Elsie Nallipogu, Junhao Wang, Linghao Zhang, Maoquan Wang, Qingwei Lin, Saravan Rajmohan, Shengyu Fu, Shilin He, Yingnong Dang, Yufan Huang, Yu Kang","submitted_at":"2025-05-29T13:09:44Z","abstract_excerpt":"The issue-resolving task, where a model generates patches to fix real-world bugs, has emerged as a critical benchmark for evaluating the capabilities of large language models (LLMs). While SWE-bench and its variants have become standard in this domain, they suffer from key limitations: they have not been updated since their initial releases, cover a narrow set of repositories, and depend heavily on manual effort for instance construction and environment setup. These factors hinder scalability and introduce risks of overfitting and data contamination. In this work, we present SWE-bench-Live, a "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.23419","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SE","submitted_at":"2025-05-29T13:09:44Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"c0abb368441d1aa01251a2bf65275f53603e481df7010658006b24529f8cb33e","abstract_canon_sha256":"62dd631ef9f12712e18c62033217fbed302047ed09abf58893e4cf48cddcd02f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:13:27.773296Z","signature_b64":"hl7SwDOnopfTrlAF+3wczzoXlYB6HDXhO9O+ALyzwYVc8zkuBgGc1kYm1XDPMXEo5iIeFtTBqNXxLIH5MiblBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cc60da4368c58fd455fc4d256e32ecc7fe30c3e0af860554d16b9212ea2b3b63","last_reissued_at":"2026-07-05T11:13:27.772807Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:13:27.772807Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SWE-bench Goes Live!","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.SE","authors_text":"Bowen Li, Chaoyun Zhang, Chengxing Xie, Dongmei Zhang, Elsie Nallipogu, Junhao Wang, Linghao Zhang, Maoquan Wang, Qingwei Lin, Saravan Rajmohan, Shengyu Fu, Shilin He, Yingnong Dang, Yufan Huang, Yu Kang","submitted_at":"2025-05-29T13:09:44Z","abstract_excerpt":"The issue-resolving task, where a model generates patches to fix real-world bugs, has emerged as a critical benchmark for evaluating the capabilities of large language models (LLMs). While SWE-bench and its variants have become standard in this domain, they suffer from key limitations: they have not been updated since their initial releases, cover a narrow set of repositories, and depend heavily on manual effort for instance construction and environment setup. These factors hinder scalability and introduce risks of overfitting and data contamination. In this work, we present SWE-bench-Live, a "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.23419","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.23419/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.23419","created_at":"2026-07-05T11:13:27.772867+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.23419v2","created_at":"2026-07-05T11:13:27.772867+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.23419","created_at":"2026-07-05T11:13:27.772867+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZRQNUQ3IYWH5","created_at":"2026-07-05T11:13:27.772867+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZRQNUQ3IYWH5IVP4","created_at":"2026-07-05T11:13:27.772867+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZRQNUQ3I","created_at":"2026-07-05T11:13:27.772867+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":22,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.07946","citing_title":"DeepSWE: Measuring Frontier Coding Agents on Original, Long-Horizon Engineering Tasks","ref_index":14,"is_internal_anchor":true},{"citing_arxiv_id":"2607.06065","citing_title":"SWE-Review: Closing the Loop on Issue Resolution with Agentic Code Review","ref_index":45,"is_internal_anchor":true},{"citing_arxiv_id":"2604.19741","citing_title":"CityRAG: Stepping Into a City via Spatially-Grounded Video Generation","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25514","citing_title":"Unlocking Model Potentials Through Adaptive Multi-Agent Scaffolding for Efficient Issue Resolution","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18284","citing_title":"Breaking the Solver Bottleneck: Training Task Generators at the Learnable Frontier","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07297","citing_title":"SWE-Explore: Benchmarking How Coding Agents Explore Repositories","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00053","citing_title":"SWE-Router: Routing in Multi-turn Agentic Software Engineering Tasks","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26177","citing_title":"RepoMirage: Probing Repository Context Reasoning in Code Agents with Perturbations","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21930","citing_title":"PITMuS: A Tool for Automated Bug Dataset Generation via Source-Level Mutant Reconstruction","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21996","citing_title":"From Patches to Trajectories: Privileged Process Supervision for Software-Engineering Agents","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2511.00780","citing_title":"Can Language Models Go Beyond Coding? Assessing the Capability of Language Models to Build Real-World Systems","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2512.18470","citing_title":"SWE-EVO: Benchmarking Coding Agents in Long-Horizon Software Evolution Scenarios","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2602.18571","citing_title":"Debug2Fix: Can Interactive Debugging Help Coding Agents Fix More Bugs?","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12925","citing_title":"AgentLens: Revealing The Lucky Pass Problem in SWE-Agent Evaluation","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13139","citing_title":"SWE-Cycle: Benchmarking Code Agents across the Complete Issue Resolution Cycle","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2509.16941","citing_title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25727","citing_title":"Toward Scalable Terminal Task Synthesis via Skill Graphs","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04637","citing_title":"SWE-WebDevBench: Evaluating Coding Agent Application Platforms as Virtual Software Agencies","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19742","citing_title":"PlayCoder: Making LLM-Generated GUI Code Playable","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18543","citing_title":"ClawEnvKit: Automatic Environment Generation for Claw-Like Agents","ref_index":98,"is_internal_anchor":false},{"citing_arxiv_id":"2602.15763","citing_title":"GLM-5: from Vibe Coding to Agentic Engineering","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21598","citing_title":"You Don't Need Public Tests to Generate Correct Code","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZRQNUQ3IYWH5IVP4JUSW4MXMY7","json":"https://pith.science/pith/ZRQNUQ3IYWH5IVP4JUSW4MXMY7.json","graph_json":"https://pith.science/api/pith-number/ZRQNUQ3IYWH5IVP4JUSW4MXMY7/graph.json","events_json":"https://pith.science/api/pith-number/ZRQNUQ3IYWH5IVP4JUSW4MXMY7/events.json","paper":"https://pith.science/paper/ZRQNUQ3I"},"agent_actions":{"view_html":"https://pith.science/pith/ZRQNUQ3IYWH5IVP4JUSW4MXMY7","download_json":"https://pith.science/pith/ZRQNUQ3IYWH5IVP4JUSW4MXMY7.json","view_paper":"https://pith.science/paper/ZRQNUQ3I","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.23419&json=true","fetch_graph":"https://pith.science/api/pith-number/ZRQNUQ3IYWH5IVP4JUSW4MXMY7/graph.json","fetch_events":"https://pith.science/api/pith-number/ZRQNUQ3IYWH5IVP4JUSW4MXMY7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZRQNUQ3IYWH5IVP4JUSW4MXMY7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZRQNUQ3IYWH5IVP4JUSW4MXMY7/action/storage_attestation","attest_author":"https://pith.science/pith/ZRQNUQ3IYWH5IVP4JUSW4MXMY7/action/author_attestation","sign_citation":"https://pith.science/pith/ZRQNUQ3IYWH5IVP4JUSW4MXMY7/action/citation_signature","submit_replication":"https://pith.science/pith/ZRQNUQ3IYWH5IVP4JUSW4MXMY7/action/replication_record"}},"created_at":"2026-07-05T11:13:27.772867+00:00","updated_at":"2026-07-05T11:13:27.772867+00:00"}