{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:BZX57LWPA7Y4YBSQGMKBA2ETAS","short_pith_number":"pith:BZX57LWP","schema_version":"1.0","canonical_sha256":"0e6fdfaecf07f1cc06503314106893048c8b5be94392e005f849d1a0c506af3d","source":{"kind":"arxiv","id":"2607.07946","version":1},"attestation_state":"computed","paper":{"title":"DeepSWE: Measuring Frontier Coding Agents on Original, Long-Horizon Engineering Tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.SE","authors_text":"Charley Lee, Leonard Tng, Serena Ge, Wenqi Huang","submitted_at":"2026-07-08T21:45:34Z","abstract_excerpt":"DeepSWE is a benchmark of 113 original, long-horizon software engineering tasks for evaluating coding agents. Most public agentic coding benchmarks follow SWE-bench in mining merged fixes from public GitHub repositories, which creates two problems: the fixes and their discussion were likely seen during pretraining, so a high score can reflect recall rather than problem-solving; and each task is graded by the tests that shipped with its merged fix, which were written to confirm one specific fix rather than grade an arbitrary solution, so they can fail a correct alternative or pass an incomplete"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.07946","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SE","submitted_at":"2026-07-08T21:45:34Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"cbc6cfadc5bc2a1867e12fcd68b5ccfef2f6699b0597cdfa178807ff18f0a9cc","abstract_canon_sha256":"1fede8ea7be57b57e23798dc66147145983da6a055e4dd44030d79a5091d0c58"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-10T00:19:05.661488Z","signature_b64":"JOR5+u9whFWNvE5zHbXoxWXlnIHlvWrJ4ThzOhOUIfGE3B544vcjDaK3MmV9NV668PgeMykcmYWLAOEIgR8gAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0e6fdfaecf07f1cc06503314106893048c8b5be94392e005f849d1a0c506af3d","last_reissued_at":"2026-07-10T00:19:05.661040Z","signature_status":"signed_v1","first_computed_at":"2026-07-10T00:19:05.661040Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DeepSWE: Measuring Frontier Coding Agents on Original, Long-Horizon Engineering Tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.SE","authors_text":"Charley Lee, Leonard Tng, Serena Ge, Wenqi Huang","submitted_at":"2026-07-08T21:45:34Z","abstract_excerpt":"DeepSWE is a benchmark of 113 original, long-horizon software engineering tasks for evaluating coding agents. Most public agentic coding benchmarks follow SWE-bench in mining merged fixes from public GitHub repositories, which creates two problems: the fixes and their discussion were likely seen during pretraining, so a high score can reflect recall rather than problem-solving; and each task is graded by the tests that shipped with its merged fix, which were written to confirm one specific fix rather than grade an arbitrary solution, so they can fail a correct alternative or pass an incomplete"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.07946","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.07946/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.07946","created_at":"2026-07-10T00:19:05.661097+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.07946v1","created_at":"2026-07-10T00:19:05.661097+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.07946","created_at":"2026-07-10T00:19:05.661097+00:00"},{"alias_kind":"pith_short_12","alias_value":"BZX57LWPA7Y4","created_at":"2026-07-10T00:19:05.661097+00:00"},{"alias_kind":"pith_short_16","alias_value":"BZX57LWPA7Y4YBSQ","created_at":"2026-07-10T00:19:05.661097+00:00"},{"alias_kind":"pith_short_8","alias_value":"BZX57LWP","created_at":"2026-07-10T00:19:05.661097+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BZX57LWPA7Y4YBSQGMKBA2ETAS","json":"https://pith.science/pith/BZX57LWPA7Y4YBSQGMKBA2ETAS.json","graph_json":"https://pith.science/api/pith-number/BZX57LWPA7Y4YBSQGMKBA2ETAS/graph.json","events_json":"https://pith.science/api/pith-number/BZX57LWPA7Y4YBSQGMKBA2ETAS/events.json","paper":"https://pith.science/paper/BZX57LWP"},"agent_actions":{"view_html":"https://pith.science/pith/BZX57LWPA7Y4YBSQGMKBA2ETAS","download_json":"https://pith.science/pith/BZX57LWPA7Y4YBSQGMKBA2ETAS.json","view_paper":"https://pith.science/paper/BZX57LWP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.07946&json=true","fetch_graph":"https://pith.science/api/pith-number/BZX57LWPA7Y4YBSQGMKBA2ETAS/graph.json","fetch_events":"https://pith.science/api/pith-number/BZX57LWPA7Y4YBSQGMKBA2ETAS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BZX57LWPA7Y4YBSQGMKBA2ETAS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BZX57LWPA7Y4YBSQGMKBA2ETAS/action/storage_attestation","attest_author":"https://pith.science/pith/BZX57LWPA7Y4YBSQGMKBA2ETAS/action/author_attestation","sign_citation":"https://pith.science/pith/BZX57LWPA7Y4YBSQGMKBA2ETAS/action/citation_signature","submit_replication":"https://pith.science/pith/BZX57LWPA7Y4YBSQGMKBA2ETAS/action/replication_record"}},"created_at":"2026-07-10T00:19:05.661097+00:00","updated_at":"2026-07-10T00:19:05.661097+00:00"}