{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:RBJUHOFM6GVUCVCSIHGX7RVSR7","short_pith_number":"pith:RBJUHOFM","schema_version":"1.0","canonical_sha256":"885343b8acf1ab41545241cd7fc6b28ff4d3dd3e9627b21a366b34c45656b4cd","source":{"kind":"arxiv","id":"2310.06912","version":1},"attestation_state":"computed","paper":{"title":"Benchmarking Deep Learning Fuzzers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Hung Viet Pham, Nima Shiri Harzevili, Song Wang","submitted_at":"2023-10-10T18:09:16Z","abstract_excerpt":"In this work, we set out to conduct the first ground-truth empirical evaluation of state-of-the-art DL fuzzers. Specifically, we first manually created an extensive DL bug benchmark dataset, which includes 627 real-world DL bugs from TensorFlow and PyTorch libraries reported by users between 2020 and 2022. Then we run three state-of-the-art DL fuzzers, i.e., FreeFuzz, DeepRel, and DocTer, on the benchmark by following their instructions. We find that these fuzzers are unable to detect many real bugs collected in our benchmark dataset. Specifically, most (235) of the 257 applicable bugs cannot "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.06912","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2023-10-10T18:09:16Z","cross_cats_sorted":[],"title_canon_sha256":"e097e5580fb2881b6556823f958e22f1821fee6130e8d63fd3da60f48a227460","abstract_canon_sha256":"3d0c34709adc4286bd82ed97b90e48c626197233eac3c8efd34ee4d32110d746"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:59:38.097479Z","signature_b64":"8gC0JS2G3XS/YPZA6aIecKaEOILrbnVdR5Kvsn7PwW7yURunlCCb8tLZ5Sl1ofaE8bq66Ey5IwQvNUyWXgdBAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"885343b8acf1ab41545241cd7fc6b28ff4d3dd3e9627b21a366b34c45656b4cd","last_reissued_at":"2026-07-05T06:59:38.097074Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:59:38.097074Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Benchmarking Deep Learning Fuzzers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Hung Viet Pham, Nima Shiri Harzevili, Song Wang","submitted_at":"2023-10-10T18:09:16Z","abstract_excerpt":"In this work, we set out to conduct the first ground-truth empirical evaluation of state-of-the-art DL fuzzers. Specifically, we first manually created an extensive DL bug benchmark dataset, which includes 627 real-world DL bugs from TensorFlow and PyTorch libraries reported by users between 2020 and 2022. Then we run three state-of-the-art DL fuzzers, i.e., FreeFuzz, DeepRel, and DocTer, on the benchmark by following their instructions. We find that these fuzzers are unable to detect many real bugs collected in our benchmark dataset. Specifically, most (235) of the 257 applicable bugs cannot "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.06912","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.06912/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.06912","created_at":"2026-07-05T06:59:38.097132+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.06912v1","created_at":"2026-07-05T06:59:38.097132+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.06912","created_at":"2026-07-05T06:59:38.097132+00:00"},{"alias_kind":"pith_short_12","alias_value":"RBJUHOFM6GVU","created_at":"2026-07-05T06:59:38.097132+00:00"},{"alias_kind":"pith_short_16","alias_value":"RBJUHOFM6GVUCVCS","created_at":"2026-07-05T06:59:38.097132+00:00"},{"alias_kind":"pith_short_8","alias_value":"RBJUHOFM","created_at":"2026-07-05T06:59:38.097132+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20128","citing_title":"The Correctness Illusion in LLM-Generated GPU Kernels","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RBJUHOFM6GVUCVCSIHGX7RVSR7","json":"https://pith.science/pith/RBJUHOFM6GVUCVCSIHGX7RVSR7.json","graph_json":"https://pith.science/api/pith-number/RBJUHOFM6GVUCVCSIHGX7RVSR7/graph.json","events_json":"https://pith.science/api/pith-number/RBJUHOFM6GVUCVCSIHGX7RVSR7/events.json","paper":"https://pith.science/paper/RBJUHOFM"},"agent_actions":{"view_html":"https://pith.science/pith/RBJUHOFM6GVUCVCSIHGX7RVSR7","download_json":"https://pith.science/pith/RBJUHOFM6GVUCVCSIHGX7RVSR7.json","view_paper":"https://pith.science/paper/RBJUHOFM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.06912&json=true","fetch_graph":"https://pith.science/api/pith-number/RBJUHOFM6GVUCVCSIHGX7RVSR7/graph.json","fetch_events":"https://pith.science/api/pith-number/RBJUHOFM6GVUCVCSIHGX7RVSR7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RBJUHOFM6GVUCVCSIHGX7RVSR7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RBJUHOFM6GVUCVCSIHGX7RVSR7/action/storage_attestation","attest_author":"https://pith.science/pith/RBJUHOFM6GVUCVCSIHGX7RVSR7/action/author_attestation","sign_citation":"https://pith.science/pith/RBJUHOFM6GVUCVCSIHGX7RVSR7/action/citation_signature","submit_replication":"https://pith.science/pith/RBJUHOFM6GVUCVCSIHGX7RVSR7/action/replication_record"}},"created_at":"2026-07-05T06:59:38.097132+00:00","updated_at":"2026-07-05T06:59:38.097132+00:00"}