{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:Z4XGXNV2AIZF2VCPHD7NRA37ZX","short_pith_number":"pith:Z4XGXNV2","schema_version":"1.0","canonical_sha256":"cf2e6bb6ba02325d544f38fed8837fcdf42342e9c735fff11bb91c88a9d23fb4","source":{"kind":"arxiv","id":"2506.13824","version":1},"attestation_state":"computed","paper":{"title":"MLDebugging: Towards Benchmarking Code Debugging Across Multi-Library Scenarios","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SE","authors_text":"Hanjie Zhao, Jiesong Bai, Jingxuan Zhou, Jinyang Huang, Libo Qin, Min Li, Qiguang Chen, Xiachong Feng, Zihui Cheng","submitted_at":"2025-06-15T13:02:59Z","abstract_excerpt":"Code debugging is a crucial task in software engineering, which attracts increasing attention. While remarkable success has been made in the era of large language models (LLMs), current research still focuses on the simple no-library or single-library setting, ignoring the complex multi-library scenario in real-world applications. To address this limitation, we make the first attempt to introduce MLDebugging (Multi-Library Debugging), a comprehensive benchmark designed to assess debugging challenges within multi-library Python code. Specifically, MLDebugging encompasses 126 distinct Python lib"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.13824","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SE","submitted_at":"2025-06-15T13:02:59Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"2c7887c901af876ec64430fe9b9e703b43def79f0f53fcb5f7f9c2e80f0f01d7","abstract_canon_sha256":"2fa4181e8081410750f66fc4c5ef15810234f1ccc3d7eb2b910fa12c1d2377fd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:22:44.283026Z","signature_b64":"f5Wek0A9MKIJ1L8uOS9nXKMiiIKegPsEm0QezpX65hBwx33PtBW8foDRa8MFrgUwutd7rsoez1Opmwtnmj+FDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cf2e6bb6ba02325d544f38fed8837fcdf42342e9c735fff11bb91c88a9d23fb4","last_reissued_at":"2026-07-05T11:22:44.282491Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:22:44.282491Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MLDebugging: Towards Benchmarking Code Debugging Across Multi-Library Scenarios","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SE","authors_text":"Hanjie Zhao, Jiesong Bai, Jingxuan Zhou, Jinyang Huang, Libo Qin, Min Li, Qiguang Chen, Xiachong Feng, Zihui Cheng","submitted_at":"2025-06-15T13:02:59Z","abstract_excerpt":"Code debugging is a crucial task in software engineering, which attracts increasing attention. While remarkable success has been made in the era of large language models (LLMs), current research still focuses on the simple no-library or single-library setting, ignoring the complex multi-library scenario in real-world applications. To address this limitation, we make the first attempt to introduce MLDebugging (Multi-Library Debugging), a comprehensive benchmark designed to assess debugging challenges within multi-library Python code. Specifically, MLDebugging encompasses 126 distinct Python lib"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.13824","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.13824/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.13824","created_at":"2026-07-05T11:22:44.282543+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.13824v1","created_at":"2026-07-05T11:22:44.282543+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.13824","created_at":"2026-07-05T11:22:44.282543+00:00"},{"alias_kind":"pith_short_12","alias_value":"Z4XGXNV2AIZF","created_at":"2026-07-05T11:22:44.282543+00:00"},{"alias_kind":"pith_short_16","alias_value":"Z4XGXNV2AIZF2VCP","created_at":"2026-07-05T11:22:44.282543+00:00"},{"alias_kind":"pith_short_8","alias_value":"Z4XGXNV2","created_at":"2026-07-05T11:22:44.282543+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2503.09567","citing_title":"Towards Reasoning Era: A Survey of Long Chain-of-Thought for Reasoning Large Language Models","ref_index":292,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Z4XGXNV2AIZF2VCPHD7NRA37ZX","json":"https://pith.science/pith/Z4XGXNV2AIZF2VCPHD7NRA37ZX.json","graph_json":"https://pith.science/api/pith-number/Z4XGXNV2AIZF2VCPHD7NRA37ZX/graph.json","events_json":"https://pith.science/api/pith-number/Z4XGXNV2AIZF2VCPHD7NRA37ZX/events.json","paper":"https://pith.science/paper/Z4XGXNV2"},"agent_actions":{"view_html":"https://pith.science/pith/Z4XGXNV2AIZF2VCPHD7NRA37ZX","download_json":"https://pith.science/pith/Z4XGXNV2AIZF2VCPHD7NRA37ZX.json","view_paper":"https://pith.science/paper/Z4XGXNV2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.13824&json=true","fetch_graph":"https://pith.science/api/pith-number/Z4XGXNV2AIZF2VCPHD7NRA37ZX/graph.json","fetch_events":"https://pith.science/api/pith-number/Z4XGXNV2AIZF2VCPHD7NRA37ZX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Z4XGXNV2AIZF2VCPHD7NRA37ZX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Z4XGXNV2AIZF2VCPHD7NRA37ZX/action/storage_attestation","attest_author":"https://pith.science/pith/Z4XGXNV2AIZF2VCPHD7NRA37ZX/action/author_attestation","sign_citation":"https://pith.science/pith/Z4XGXNV2AIZF2VCPHD7NRA37ZX/action/citation_signature","submit_replication":"https://pith.science/pith/Z4XGXNV2AIZF2VCPHD7NRA37ZX/action/replication_record"}},"created_at":"2026-07-05T11:22:44.282543+00:00","updated_at":"2026-07-05T11:22:44.282543+00:00"}