{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:BRBMAHRISL2YRLXGCPZLMKEJ7M","short_pith_number":"pith:BRBMAHRI","schema_version":"1.0","canonical_sha256":"0c42c01e2892f588aee613f2b62889fb1dba408e0f245c610ada6e59d5eb9f26","source":{"kind":"arxiv","id":"2505.07473","version":1},"attestation_state":"computed","paper":{"title":"Web-Bench: A LLM Code Benchmark Based on Web Standards and Frameworks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Kai Xu, Xinyi Guan, YiWei Mao, ZiLong Feng","submitted_at":"2025-05-12T12:06:23Z","abstract_excerpt":"The application of large language models (LLMs) in the field of coding is evolving rapidly: from code assistants, to autonomous coding agents, and then to generating complete projects through natural language. Early LLM code benchmarks primarily focused on code generation accuracy, but these benchmarks have gradually become saturated. Benchmark saturation weakens their guiding role for LLMs. For example, HumanEval Pass@1 has reached 99.4% and MBPP 94.2%. Among various attempts to address benchmark saturation, approaches based on software engineering have stood out, but the saturation of existi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.07473","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-05-12T12:06:23Z","cross_cats_sorted":[],"title_canon_sha256":"83481171577798fc9501985ca6d16d69265659e9c641b4e883a3671ead0a75c6","abstract_canon_sha256":"1615f0de8d42190a994111b30efafebab60df7fbc9251180346e583e82a4b668"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:01:48.347050Z","signature_b64":"BkLf8jHCCeJ0keaDDple0otyv0ejp7G/YjAnk+PMGhAGWzJBVJpJcNw2HSr9+P5+mj6kbchU6sg+yvfkvVf/AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0c42c01e2892f588aee613f2b62889fb1dba408e0f245c610ada6e59d5eb9f26","last_reissued_at":"2026-07-05T11:01:48.346575Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:01:48.346575Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Web-Bench: A LLM Code Benchmark Based on Web Standards and Frameworks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Kai Xu, Xinyi Guan, YiWei Mao, ZiLong Feng","submitted_at":"2025-05-12T12:06:23Z","abstract_excerpt":"The application of large language models (LLMs) in the field of coding is evolving rapidly: from code assistants, to autonomous coding agents, and then to generating complete projects through natural language. Early LLM code benchmarks primarily focused on code generation accuracy, but these benchmarks have gradually become saturated. Benchmark saturation weakens their guiding role for LLMs. For example, HumanEval Pass@1 has reached 99.4% and MBPP 94.2%. Among various attempts to address benchmark saturation, approaches based on software engineering have stood out, but the saturation of existi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.07473","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.07473/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.07473","created_at":"2026-07-05T11:01:48.346630+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.07473v1","created_at":"2026-07-05T11:01:48.346630+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.07473","created_at":"2026-07-05T11:01:48.346630+00:00"},{"alias_kind":"pith_short_12","alias_value":"BRBMAHRISL2Y","created_at":"2026-07-05T11:01:48.346630+00:00"},{"alias_kind":"pith_short_16","alias_value":"BRBMAHRISL2YRLXG","created_at":"2026-07-05T11:01:48.346630+00:00"},{"alias_kind":"pith_short_8","alias_value":"BRBMAHRI","created_at":"2026-07-05T11:01:48.346630+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07480","citing_title":"Biased or Personalized? The Impact of Personal Information on AI-driven Development","ref_index":71,"is_internal_anchor":true},{"citing_arxiv_id":"2606.05920","citing_title":"Asuka-Bench: Benchmarking Code Agents on Underspecified User Intent and Multi-Round Refinement","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01869","citing_title":"WorldCoder-Bench: Benchmarking Physically Grounded 3D World Synthesis","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00154","citing_title":"Benchmarking Multimodal LLMs on Code Generation for Complex Interactive Webpages","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30000","citing_title":"Cookie-Bench: Continuous On-screen Key Interaction Evaluation for Web Generation","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00750","citing_title":"I-WebGenBench : Evaluating Interactivity in LLM-Generated Scientific Web Applications","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15222","citing_title":"PerfCodeBench: Benchmarking LLMs for System-Level High-Performance Code Optimization","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2601.16456","citing_title":"RubberDuckBench: A Benchmark for AI Coding Assistants","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19750","citing_title":"Coding with Eyes: Visual Feedback Unlocks Reliable GUI Code Generating and Debugging","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BRBMAHRISL2YRLXGCPZLMKEJ7M","json":"https://pith.science/pith/BRBMAHRISL2YRLXGCPZLMKEJ7M.json","graph_json":"https://pith.science/api/pith-number/BRBMAHRISL2YRLXGCPZLMKEJ7M/graph.json","events_json":"https://pith.science/api/pith-number/BRBMAHRISL2YRLXGCPZLMKEJ7M/events.json","paper":"https://pith.science/paper/BRBMAHRI"},"agent_actions":{"view_html":"https://pith.science/pith/BRBMAHRISL2YRLXGCPZLMKEJ7M","download_json":"https://pith.science/pith/BRBMAHRISL2YRLXGCPZLMKEJ7M.json","view_paper":"https://pith.science/paper/BRBMAHRI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.07473&json=true","fetch_graph":"https://pith.science/api/pith-number/BRBMAHRISL2YRLXGCPZLMKEJ7M/graph.json","fetch_events":"https://pith.science/api/pith-number/BRBMAHRISL2YRLXGCPZLMKEJ7M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BRBMAHRISL2YRLXGCPZLMKEJ7M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BRBMAHRISL2YRLXGCPZLMKEJ7M/action/storage_attestation","attest_author":"https://pith.science/pith/BRBMAHRISL2YRLXGCPZLMKEJ7M/action/author_attestation","sign_citation":"https://pith.science/pith/BRBMAHRISL2YRLXGCPZLMKEJ7M/action/citation_signature","submit_replication":"https://pith.science/pith/BRBMAHRISL2YRLXGCPZLMKEJ7M/action/replication_record"}},"created_at":"2026-07-05T11:01:48.346630+00:00","updated_at":"2026-07-05T11:01:48.346630+00:00"}