{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XTHFAQSO6OOYG6PO6XTYRLYGP3","short_pith_number":"pith:XTHFAQSO","schema_version":"1.0","canonical_sha256":"bcce50424ef39d8379eef5e788af067eff5092490c3c492b6092900260f60262","source":{"kind":"arxiv","id":"2408.11053","version":2},"attestation_state":"computed","paper":{"title":"Revisiting VerilogEval: A Year of Improvements in Large-Language Models for Hardware Code Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.AR","authors_text":"Brucek Khailany, Christopher Batten, Haoxing Ren, Mingjie Liu, Nathaniel Pinckney","submitted_at":"2024-08-20T17:58:56Z","abstract_excerpt":"The application of large-language models (LLMs) to digital hardware code generation is an emerging field, with most LLMs primarily trained on natural language and software code. Hardware code like Verilog constitutes a small portion of training data, and few hardware benchmarks exist. The open-source VerilogEval benchmark, released in November 2023, provided a consistent evaluation framework for LLMs on code completion tasks. Since then, both commercial and open models have seen significant development.\n  In this work, we evaluate new commercial and open models since VerilogEval's original rel"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.11053","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AR","submitted_at":"2024-08-20T17:58:56Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"7a2b4b1ca5131be2b6a45b6c5cd3cd35bf4e7fcd48905f32167edfe7bf938a2a","abstract_canon_sha256":"db3ad56d82b256ee06d0465c2bde0327c4416c50a37bb70affba367f3f7d3729"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:09:07.680506Z","signature_b64":"6CfHSD8eWb5gsS+pIy5iAFy1LViD53ta+zoGRvi9AwVHWCQ7vi3OsYnYF2bz5Ws99isiqO5nZVtKF2WgQi67Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bcce50424ef39d8379eef5e788af067eff5092490c3c492b6092900260f60262","last_reissued_at":"2026-07-05T10:09:07.679920Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:09:07.679920Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Revisiting VerilogEval: A Year of Improvements in Large-Language Models for Hardware Code Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.AR","authors_text":"Brucek Khailany, Christopher Batten, Haoxing Ren, Mingjie Liu, Nathaniel Pinckney","submitted_at":"2024-08-20T17:58:56Z","abstract_excerpt":"The application of large-language models (LLMs) to digital hardware code generation is an emerging field, with most LLMs primarily trained on natural language and software code. Hardware code like Verilog constitutes a small portion of training data, and few hardware benchmarks exist. The open-source VerilogEval benchmark, released in November 2023, provided a consistent evaluation framework for LLMs on code completion tasks. Since then, both commercial and open models have seen significant development.\n  In this work, we evaluate new commercial and open models since VerilogEval's original rel"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.11053","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.11053/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.11053","created_at":"2026-07-05T10:09:07.679981+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.11053v2","created_at":"2026-07-05T10:09:07.679981+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.11053","created_at":"2026-07-05T10:09:07.679981+00:00"},{"alias_kind":"pith_short_12","alias_value":"XTHFAQSO6OOY","created_at":"2026-07-05T10:09:07.679981+00:00"},{"alias_kind":"pith_short_16","alias_value":"XTHFAQSO6OOYG6PO","created_at":"2026-07-05T10:09:07.679981+00:00"},{"alias_kind":"pith_short_8","alias_value":"XTHFAQSO","created_at":"2026-07-05T10:09:07.679981+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17253","citing_title":"PDAGENT-BENCH: Characterizing, Grounding, and Architecting LLM Agents for VLSI Physical Design","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08976","citing_title":"RTL-BenchLS: A Large-Scale Benchmark for RTL Reasoning and Generation with Large Language Models","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12857","citing_title":"ChipMATE: Multi-Agent Training via Reinforcement Learning for Enhanced RTL Generation","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26097","citing_title":"Forgetting in Language Models: Capacity, Optimization, and Self-Generated Replay","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26498","citing_title":"Verilog-Evolve: Feedback-Driven and Skill-Evolving Verilog Generation","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23355","citing_title":"LEGO: An LLM Skill-Based Front-End Design Generation Platform","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15537","citing_title":"RTL-BenchMT: Dynamic Maintenance of RTL Generation Benchmark Through Agent-Assisted Analysis and Revision","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23355","citing_title":"LEGO: An LLM Skill-Based Front-End Design Generation Platform","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XTHFAQSO6OOYG6PO6XTYRLYGP3","json":"https://pith.science/pith/XTHFAQSO6OOYG6PO6XTYRLYGP3.json","graph_json":"https://pith.science/api/pith-number/XTHFAQSO6OOYG6PO6XTYRLYGP3/graph.json","events_json":"https://pith.science/api/pith-number/XTHFAQSO6OOYG6PO6XTYRLYGP3/events.json","paper":"https://pith.science/paper/XTHFAQSO"},"agent_actions":{"view_html":"https://pith.science/pith/XTHFAQSO6OOYG6PO6XTYRLYGP3","download_json":"https://pith.science/pith/XTHFAQSO6OOYG6PO6XTYRLYGP3.json","view_paper":"https://pith.science/paper/XTHFAQSO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.11053&json=true","fetch_graph":"https://pith.science/api/pith-number/XTHFAQSO6OOYG6PO6XTYRLYGP3/graph.json","fetch_events":"https://pith.science/api/pith-number/XTHFAQSO6OOYG6PO6XTYRLYGP3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XTHFAQSO6OOYG6PO6XTYRLYGP3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XTHFAQSO6OOYG6PO6XTYRLYGP3/action/storage_attestation","attest_author":"https://pith.science/pith/XTHFAQSO6OOYG6PO6XTYRLYGP3/action/author_attestation","sign_citation":"https://pith.science/pith/XTHFAQSO6OOYG6PO6XTYRLYGP3/action/citation_signature","submit_replication":"https://pith.science/pith/XTHFAQSO6OOYG6PO6XTYRLYGP3/action/replication_record"}},"created_at":"2026-07-05T10:09:07.679981+00:00","updated_at":"2026-07-05T10:09:07.679981+00:00"}