{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:EOH2FYYDP2YBOTL37ZM2O3J2NH","short_pith_number":"pith:EOH2FYYD","schema_version":"1.0","canonical_sha256":"238fa2e3037eb0174d7bfe59a76d3a69fdcbef3800b69ebd3eb59be7989d61f6","source":{"kind":"arxiv","id":"2507.18130","version":3},"attestation_state":"computed","paper":{"title":"NoCode-bench: A Benchmark for Evaluating Natural Language-Driven Feature Addition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Jialun Cao, Le Deng, Michael Pradel, Zhonghao Jiang, Zhongxin Liu","submitted_at":"2025-07-24T06:38:19Z","abstract_excerpt":"Natural language-driven no-code development allows users to specify software functionality using natural language (NL) instead of editing source code, promising increased productivity and democratized development. Large language models (LLMs) show potential in enabling this paradigm. In this context, software documentation acts as an NL specification for functionality. This work introduces NoCode-bench, a benchmark designed to evaluate LLMs on real-world NL-driven feature addition tasks, consisting of 634 tasks across 10 projects and 114k code changes. Each task pairs documentation updates wit"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.18130","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SE","submitted_at":"2025-07-24T06:38:19Z","cross_cats_sorted":[],"title_canon_sha256":"04bedfa81e5e40491ee004ec195a3eb821405cd473050edbffd3c1ee77a18c19","abstract_canon_sha256":"718889c4bdbafafe4125460cea0d772850ad75ade389d65b805d19920c937e15"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:55:20.607127Z","signature_b64":"cSMqLWZkikHY/gc2DUtkpgR3TSJnmaax06Ait2VDuCpvlAmpfqhBhMKhDVLDeL5VmuCQmSExbOQXILpb80J0Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"238fa2e3037eb0174d7bfe59a76d3a69fdcbef3800b69ebd3eb59be7989d61f6","last_reissued_at":"2026-07-05T11:55:20.606662Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:55:20.606662Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"NoCode-bench: A Benchmark for Evaluating Natural Language-Driven Feature Addition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Jialun Cao, Le Deng, Michael Pradel, Zhonghao Jiang, Zhongxin Liu","submitted_at":"2025-07-24T06:38:19Z","abstract_excerpt":"Natural language-driven no-code development allows users to specify software functionality using natural language (NL) instead of editing source code, promising increased productivity and democratized development. Large language models (LLMs) show potential in enabling this paradigm. In this context, software documentation acts as an NL specification for functionality. This work introduces NoCode-bench, a benchmark designed to evaluate LLMs on real-world NL-driven feature addition tasks, consisting of 634 tasks across 10 projects and 114k code changes. Each task pairs documentation updates wit"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.18130","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.18130/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.18130","created_at":"2026-07-05T11:55:20.606718+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.18130v3","created_at":"2026-07-05T11:55:20.606718+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.18130","created_at":"2026-07-05T11:55:20.606718+00:00"},{"alias_kind":"pith_short_12","alias_value":"EOH2FYYDP2YB","created_at":"2026-07-05T11:55:20.606718+00:00"},{"alias_kind":"pith_short_16","alias_value":"EOH2FYYDP2YBOTL3","created_at":"2026-07-05T11:55:20.606718+00:00"},{"alias_kind":"pith_short_8","alias_value":"EOH2FYYD","created_at":"2026-07-05T11:55:20.606718+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":5,"sample":[{"citing_arxiv_id":"2605.25356","citing_title":"Names Are All You Need: Effective and Safe Regression Test Selection for Python","ref_index":20,"is_internal_anchor":true},{"citing_arxiv_id":"2605.04320","citing_title":"Reproduction Test Generation for Java SWE Issues","ref_index":12,"is_internal_anchor":true},{"citing_arxiv_id":"2604.11270","citing_title":"Evaluating LLM Agents on Automated Software Analysis Tasks","ref_index":13,"is_internal_anchor":true},{"citing_arxiv_id":"2605.04320","citing_title":"Reproduction Test Generation for Java SWE Issues","ref_index":12,"is_internal_anchor":true},{"citing_arxiv_id":"2604.16021","citing_title":"Neurosymbolic Repo-level Code Localization","ref_index":5,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EOH2FYYDP2YBOTL37ZM2O3J2NH","json":"https://pith.science/pith/EOH2FYYDP2YBOTL37ZM2O3J2NH.json","graph_json":"https://pith.science/api/pith-number/EOH2FYYDP2YBOTL37ZM2O3J2NH/graph.json","events_json":"https://pith.science/api/pith-number/EOH2FYYDP2YBOTL37ZM2O3J2NH/events.json","paper":"https://pith.science/paper/EOH2FYYD"},"agent_actions":{"view_html":"https://pith.science/pith/EOH2FYYDP2YBOTL37ZM2O3J2NH","download_json":"https://pith.science/pith/EOH2FYYDP2YBOTL37ZM2O3J2NH.json","view_paper":"https://pith.science/paper/EOH2FYYD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.18130&json=true","fetch_graph":"https://pith.science/api/pith-number/EOH2FYYDP2YBOTL37ZM2O3J2NH/graph.json","fetch_events":"https://pith.science/api/pith-number/EOH2FYYDP2YBOTL37ZM2O3J2NH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EOH2FYYDP2YBOTL37ZM2O3J2NH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EOH2FYYDP2YBOTL37ZM2O3J2NH/action/storage_attestation","attest_author":"https://pith.science/pith/EOH2FYYDP2YBOTL37ZM2O3J2NH/action/author_attestation","sign_citation":"https://pith.science/pith/EOH2FYYDP2YBOTL37ZM2O3J2NH/action/citation_signature","submit_replication":"https://pith.science/pith/EOH2FYYDP2YBOTL37ZM2O3J2NH/action/replication_record"}},"created_at":"2026-07-05T11:55:20.606718+00:00","updated_at":"2026-07-05T11:55:20.606718+00:00"}