{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:QQRYZEWTECP4D6S3F3WQBSTQD7","short_pith_number":"pith:QQRYZEWT","schema_version":"1.0","canonical_sha256":"84238c92d3209fc1fa5b2eed00ca701fff5bd47290ef87c194ac20b255a42b9e","source":{"kind":"arxiv","id":"2607.29283","version":1},"attestation_state":"computed","paper":{"title":"RTLCurator: Label-Efficient Data Curation for RTL Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AR","authors_text":"Cangyuan Li, Haoyu Gao, Kun Wang, Siyang Cai, Wenjing Chang, Ying Wang, Yinhe Han","submitted_at":"2026-07-31T10:55:01Z","abstract_excerpt":"Training large language models (LLMs) to write register-transfer level (RTL) requires large corpora of paired specifications and code, and such data is scarce enough that most public corpora are now synthesized. Synthesis provides scale but not correctness, and in two widely used RTL datasets only 24.4% and 53.5% of pairs pass generated functional tests. This raises the question of how much of such a corpus to keep and which part of it. Correctness alone is a poor answer. A pair that misbehaves in one corner case still shows valid syntax and interface conventions, and complex sequential design"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.29283","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AR","submitted_at":"2026-07-31T10:55:01Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"842c2f5c666f3feafafafc1185da253dc014b5462cc26a4f4a951a126b3a1348","abstract_canon_sha256":"3e5fd5ce443fc87b012b3f5035b4f98f6edd5e9b6a5dc5f49084af01344f5dee"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-03T01:23:38.747796Z","signature_b64":"ZMpERBJu6ihTqFH50fC0mC+f0iqAcsJEuCZdh4QGDvVYqA8r/qOJPPszlgQIUTgO0zBjywyOFyy0KvG32yzHBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"84238c92d3209fc1fa5b2eed00ca701fff5bd47290ef87c194ac20b255a42b9e","last_reissued_at":"2026-08-03T01:23:38.746159Z","signature_status":"signed_v1","first_computed_at":"2026-08-03T01:23:38.746159Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RTLCurator: Label-Efficient Data Curation for RTL Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AR","authors_text":"Cangyuan Li, Haoyu Gao, Kun Wang, Siyang Cai, Wenjing Chang, Ying Wang, Yinhe Han","submitted_at":"2026-07-31T10:55:01Z","abstract_excerpt":"Training large language models (LLMs) to write register-transfer level (RTL) requires large corpora of paired specifications and code, and such data is scarce enough that most public corpora are now synthesized. Synthesis provides scale but not correctness, and in two widely used RTL datasets only 24.4% and 53.5% of pairs pass generated functional tests. This raises the question of how much of such a corpus to keep and which part of it. Correctness alone is a poor answer. A pair that misbehaves in one corner case still shows valid syntax and interface conventions, and complex sequential design"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.29283","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.29283/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.29283","created_at":"2026-08-03T01:23:38.747085+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.29283v1","created_at":"2026-08-03T01:23:38.747085+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.29283","created_at":"2026-08-03T01:23:38.747085+00:00"},{"alias_kind":"pith_short_12","alias_value":"QQRYZEWTECP4","created_at":"2026-08-03T01:23:38.747085+00:00"},{"alias_kind":"pith_short_16","alias_value":"QQRYZEWTECP4D6S3","created_at":"2026-08-03T01:23:38.747085+00:00"},{"alias_kind":"pith_short_8","alias_value":"QQRYZEWT","created_at":"2026-08-03T01:23:38.747085+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QQRYZEWTECP4D6S3F3WQBSTQD7","json":"https://pith.science/pith/QQRYZEWTECP4D6S3F3WQBSTQD7.json","graph_json":"https://pith.science/api/pith-number/QQRYZEWTECP4D6S3F3WQBSTQD7/graph.json","events_json":"https://pith.science/api/pith-number/QQRYZEWTECP4D6S3F3WQBSTQD7/events.json","paper":"https://pith.science/paper/QQRYZEWT"},"agent_actions":{"view_html":"https://pith.science/pith/QQRYZEWTECP4D6S3F3WQBSTQD7","download_json":"https://pith.science/pith/QQRYZEWTECP4D6S3F3WQBSTQD7.json","view_paper":"https://pith.science/paper/QQRYZEWT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.29283&json=true","fetch_graph":"https://pith.science/api/pith-number/QQRYZEWTECP4D6S3F3WQBSTQD7/graph.json","fetch_events":"https://pith.science/api/pith-number/QQRYZEWTECP4D6S3F3WQBSTQD7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QQRYZEWTECP4D6S3F3WQBSTQD7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QQRYZEWTECP4D6S3F3WQBSTQD7/action/storage_attestation","attest_author":"https://pith.science/pith/QQRYZEWTECP4D6S3F3WQBSTQD7/action/author_attestation","sign_citation":"https://pith.science/pith/QQRYZEWTECP4D6S3F3WQBSTQD7/action/citation_signature","submit_replication":"https://pith.science/pith/QQRYZEWTECP4D6S3F3WQBSTQD7/action/replication_record"}},"created_at":"2026-08-03T01:23:38.747085+00:00","updated_at":"2026-08-03T01:23:38.747085+00:00"}