{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:654EYS7ERDOJM7X3WH2JYA6XM7","short_pith_number":"pith:654EYS7E","schema_version":"1.0","canonical_sha256":"f7784c4be488dc967efbb1f49c03d767d4e9cffac8ca5b3cbe29077138b97d79","source":{"kind":"arxiv","id":"2505.12331","version":2},"attestation_state":"computed","paper":{"title":"OSS-Bench: Benchmark Generator for Coding LLMs","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.SE","authors_text":"Roland Yap, Yuancheng Jiang, Zhenkai Liang","submitted_at":"2025-05-18T09:53:51Z","abstract_excerpt":"In light of the rapid adoption of AI coding assistants, LLM-assisted development has become increasingly prevalent, creating an urgent need for robust evaluation of generated code quality. Existing benchmarks often require extensive manual effort to create static datasets, rely on indirect or insufficiently challenging tasks, depend on non-scalable ground truth, or neglect critical low-level security evaluations, particularly memory-safety issues. In this work, we introduce OSS-Bench, a benchmark generator that automatically constructs large-scale, live evaluation tasks from real-world open-so"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.12331","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.SE","submitted_at":"2025-05-18T09:53:51Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"95da90041d6b3d25e17d63492368c51407b08878abc7056bfaf215427764b08e","abstract_canon_sha256":"7967828f73286bf6c9be1559d3e42de5c78787185e4392e12f66db524154b904"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:05:45.281002Z","signature_b64":"hAX/IVYiOcDz3xt7hi3LvkfUEqP8bIAc0UId3MEiIyiI0dIWgmh1f+Lxg3UKKHkCuVskjVA9QrFJST+149aQBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f7784c4be488dc967efbb1f49c03d767d4e9cffac8ca5b3cbe29077138b97d79","last_reissued_at":"2026-07-05T11:05:45.280531Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:05:45.280531Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OSS-Bench: Benchmark Generator for Coding LLMs","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.SE","authors_text":"Roland Yap, Yuancheng Jiang, Zhenkai Liang","submitted_at":"2025-05-18T09:53:51Z","abstract_excerpt":"In light of the rapid adoption of AI coding assistants, LLM-assisted development has become increasingly prevalent, creating an urgent need for robust evaluation of generated code quality. Existing benchmarks often require extensive manual effort to create static datasets, rely on indirect or insufficiently challenging tasks, depend on non-scalable ground truth, or neglect critical low-level security evaluations, particularly memory-safety issues. In this work, we introduce OSS-Bench, a benchmark generator that automatically constructs large-scale, live evaluation tasks from real-world open-so"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.12331","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.12331/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.12331","created_at":"2026-07-05T11:05:45.280587+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.12331v2","created_at":"2026-07-05T11:05:45.280587+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.12331","created_at":"2026-07-05T11:05:45.280587+00:00"},{"alias_kind":"pith_short_12","alias_value":"654EYS7ERDOJ","created_at":"2026-07-05T11:05:45.280587+00:00"},{"alias_kind":"pith_short_16","alias_value":"654EYS7ERDOJM7X3","created_at":"2026-07-05T11:05:45.280587+00:00"},{"alias_kind":"pith_short_8","alias_value":"654EYS7E","created_at":"2026-07-05T11:05:45.280587+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/654EYS7ERDOJM7X3WH2JYA6XM7","json":"https://pith.science/pith/654EYS7ERDOJM7X3WH2JYA6XM7.json","graph_json":"https://pith.science/api/pith-number/654EYS7ERDOJM7X3WH2JYA6XM7/graph.json","events_json":"https://pith.science/api/pith-number/654EYS7ERDOJM7X3WH2JYA6XM7/events.json","paper":"https://pith.science/paper/654EYS7E"},"agent_actions":{"view_html":"https://pith.science/pith/654EYS7ERDOJM7X3WH2JYA6XM7","download_json":"https://pith.science/pith/654EYS7ERDOJM7X3WH2JYA6XM7.json","view_paper":"https://pith.science/paper/654EYS7E","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.12331&json=true","fetch_graph":"https://pith.science/api/pith-number/654EYS7ERDOJM7X3WH2JYA6XM7/graph.json","fetch_events":"https://pith.science/api/pith-number/654EYS7ERDOJM7X3WH2JYA6XM7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/654EYS7ERDOJM7X3WH2JYA6XM7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/654EYS7ERDOJM7X3WH2JYA6XM7/action/storage_attestation","attest_author":"https://pith.science/pith/654EYS7ERDOJM7X3WH2JYA6XM7/action/author_attestation","sign_citation":"https://pith.science/pith/654EYS7ERDOJM7X3WH2JYA6XM7/action/citation_signature","submit_replication":"https://pith.science/pith/654EYS7ERDOJM7X3WH2JYA6XM7/action/replication_record"}},"created_at":"2026-07-05T11:05:45.280587+00:00","updated_at":"2026-07-05T11:05:45.280587+00:00"}