{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:UV3OZGZHROAE2PMTVQS6NWSHJB","short_pith_number":"pith:UV3OZGZH","schema_version":"1.0","canonical_sha256":"a576ec9b278b804d3d93ac25e6da4748659ae9e1f69288b8c65b5d7723b333e8","source":{"kind":"arxiv","id":"2505.09027","version":1},"attestation_state":"computed","paper":{"title":"Tests as Prompt: A Test-Driven-Development Benchmark for LLM Code Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SE","authors_text":"Yi Cui","submitted_at":"2025-05-13T23:47:12Z","abstract_excerpt":"We introduce WebApp1K, a novel benchmark for evaluating large language models (LLMs) in test-driven development (TDD) tasks, where test cases serve as both prompt and verification for code generation. Unlike traditional approaches relying on natural language prompts, our benchmark emphasizes the ability of LLMs to interpret and implement functionality directly from test cases, reflecting real-world software development practices. Comprising 1000 diverse challenges across 20 application domains, the benchmark evaluates LLMs on their ability to generate compact, functional code under the constra"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.09027","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2025-05-13T23:47:12Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"b6430a3dd913ad6eb87d1851c0f67db96e5448f8d354d9ada2e363ea596c11d2","abstract_canon_sha256":"4c12011e727caad319623d6eb3a01ae7f38e0d4d954519686aa72433395d1fe5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:02:49.075768Z","signature_b64":"vIh/H1Ohs8P8KQZblKLgMftcm4EBReLAGEhPBvKwL3eXM5aMdVxhlPbHafs4YUo+K7OKs0qbYsHDuCNdZEGSAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a576ec9b278b804d3d93ac25e6da4748659ae9e1f69288b8c65b5d7723b333e8","last_reissued_at":"2026-07-05T11:02:49.075328Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:02:49.075328Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Tests as Prompt: A Test-Driven-Development Benchmark for LLM Code Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SE","authors_text":"Yi Cui","submitted_at":"2025-05-13T23:47:12Z","abstract_excerpt":"We introduce WebApp1K, a novel benchmark for evaluating large language models (LLMs) in test-driven development (TDD) tasks, where test cases serve as both prompt and verification for code generation. Unlike traditional approaches relying on natural language prompts, our benchmark emphasizes the ability of LLMs to interpret and implement functionality directly from test cases, reflecting real-world software development practices. Comprising 1000 diverse challenges across 20 application domains, the benchmark evaluates LLMs on their ability to generate compact, functional code under the constra"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.09027","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.09027/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.09027","created_at":"2026-07-05T11:02:49.075390+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.09027v1","created_at":"2026-07-05T11:02:49.075390+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.09027","created_at":"2026-07-05T11:02:49.075390+00:00"},{"alias_kind":"pith_short_12","alias_value":"UV3OZGZHROAE","created_at":"2026-07-05T11:02:49.075390+00:00"},{"alias_kind":"pith_short_16","alias_value":"UV3OZGZHROAE2PMT","created_at":"2026-07-05T11:02:49.075390+00:00"},{"alias_kind":"pith_short_8","alias_value":"UV3OZGZH","created_at":"2026-07-05T11:02:49.075390+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08135","citing_title":"TICoder: A Repository-Level Code Generation Framework with Test-Driven Planning and Implementation-Aware Reuse","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26615","citing_title":"TDD Governance for Multi-Agent Code Generation via Prompt Engineering","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UV3OZGZHROAE2PMTVQS6NWSHJB","json":"https://pith.science/pith/UV3OZGZHROAE2PMTVQS6NWSHJB.json","graph_json":"https://pith.science/api/pith-number/UV3OZGZHROAE2PMTVQS6NWSHJB/graph.json","events_json":"https://pith.science/api/pith-number/UV3OZGZHROAE2PMTVQS6NWSHJB/events.json","paper":"https://pith.science/paper/UV3OZGZH"},"agent_actions":{"view_html":"https://pith.science/pith/UV3OZGZHROAE2PMTVQS6NWSHJB","download_json":"https://pith.science/pith/UV3OZGZHROAE2PMTVQS6NWSHJB.json","view_paper":"https://pith.science/paper/UV3OZGZH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.09027&json=true","fetch_graph":"https://pith.science/api/pith-number/UV3OZGZHROAE2PMTVQS6NWSHJB/graph.json","fetch_events":"https://pith.science/api/pith-number/UV3OZGZHROAE2PMTVQS6NWSHJB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UV3OZGZHROAE2PMTVQS6NWSHJB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UV3OZGZHROAE2PMTVQS6NWSHJB/action/storage_attestation","attest_author":"https://pith.science/pith/UV3OZGZHROAE2PMTVQS6NWSHJB/action/author_attestation","sign_citation":"https://pith.science/pith/UV3OZGZHROAE2PMTVQS6NWSHJB/action/citation_signature","submit_replication":"https://pith.science/pith/UV3OZGZHROAE2PMTVQS6NWSHJB/action/replication_record"}},"created_at":"2026-07-05T11:02:49.075390+00:00","updated_at":"2026-07-05T11:02:49.075390+00:00"}