{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:5J4VGTBV3DUQP5IZKIIUZLS3IN","short_pith_number":"pith:5J4VGTBV","schema_version":"1.0","canonical_sha256":"ea79534c35d8e907f51952114cae5b434ee541df5b8e3ee21986e5d99ec5e46d","source":{"kind":"arxiv","id":"2505.24324","version":1},"attestation_state":"computed","paper":{"title":"SwiftEval: Developing a Language-Specific Benchmark for LLM-generated Code Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.PL","cs.SE"],"primary_cat":"cs.LG","authors_text":"Ivan Petrukha, Nataliia Stulova, Yana Kurliak","submitted_at":"2025-05-30T08:06:30Z","abstract_excerpt":"In recent years, large language models (LLMs) have showcased significant advancements in code generation. However, most evaluation benchmarks are primarily oriented towards Python, making it difficult to evaluate other programming languages, such as Swift, with high quality. By examining widely established multilingual benchmarks like HumanEval-XL and MultiPL-E, we identified critical issues specific to their Swift components, making them insufficient or even irrelevant for assessing LLM coding capabilities on Swift. Unlike these existing approaches, which prioritize rapid scaling and generali"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.24324","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-05-30T08:06:30Z","cross_cats_sorted":["cs.CL","cs.PL","cs.SE"],"title_canon_sha256":"9d191ae85fb3e580275ea91a6413519ad98423bd20eeba6522c106ffed7c54b9","abstract_canon_sha256":"a8a30f47b89b420ac532cc012a4d40036b3e7197b3a944c19276467298919f55"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:12:46.863128Z","signature_b64":"ymgcecZGTC2s1xsF20Ppm7h3GCsJxutA5XhMX8H6nio2VrglvhCmR0xUakGRF5ImzLWZqEYoI3lkh6KvvaUmDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ea79534c35d8e907f51952114cae5b434ee541df5b8e3ee21986e5d99ec5e46d","last_reissued_at":"2026-07-05T11:12:46.862103Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:12:46.862103Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SwiftEval: Developing a Language-Specific Benchmark for LLM-generated Code Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.PL","cs.SE"],"primary_cat":"cs.LG","authors_text":"Ivan Petrukha, Nataliia Stulova, Yana Kurliak","submitted_at":"2025-05-30T08:06:30Z","abstract_excerpt":"In recent years, large language models (LLMs) have showcased significant advancements in code generation. However, most evaluation benchmarks are primarily oriented towards Python, making it difficult to evaluate other programming languages, such as Swift, with high quality. By examining widely established multilingual benchmarks like HumanEval-XL and MultiPL-E, we identified critical issues specific to their Swift components, making them insufficient or even irrelevant for assessing LLM coding capabilities on Swift. Unlike these existing approaches, which prioritize rapid scaling and generali"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.24324","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.24324/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.24324","created_at":"2026-07-05T11:12:46.862650+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.24324v1","created_at":"2026-07-05T11:12:46.862650+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.24324","created_at":"2026-07-05T11:12:46.862650+00:00"},{"alias_kind":"pith_short_12","alias_value":"5J4VGTBV3DUQ","created_at":"2026-07-05T11:12:46.862650+00:00"},{"alias_kind":"pith_short_16","alias_value":"5J4VGTBV3DUQP5IZ","created_at":"2026-07-05T11:12:46.862650+00:00"},{"alias_kind":"pith_short_8","alias_value":"5J4VGTBV","created_at":"2026-07-05T11:12:46.862650+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5J4VGTBV3DUQP5IZKIIUZLS3IN","json":"https://pith.science/pith/5J4VGTBV3DUQP5IZKIIUZLS3IN.json","graph_json":"https://pith.science/api/pith-number/5J4VGTBV3DUQP5IZKIIUZLS3IN/graph.json","events_json":"https://pith.science/api/pith-number/5J4VGTBV3DUQP5IZKIIUZLS3IN/events.json","paper":"https://pith.science/paper/5J4VGTBV"},"agent_actions":{"view_html":"https://pith.science/pith/5J4VGTBV3DUQP5IZKIIUZLS3IN","download_json":"https://pith.science/pith/5J4VGTBV3DUQP5IZKIIUZLS3IN.json","view_paper":"https://pith.science/paper/5J4VGTBV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.24324&json=true","fetch_graph":"https://pith.science/api/pith-number/5J4VGTBV3DUQP5IZKIIUZLS3IN/graph.json","fetch_events":"https://pith.science/api/pith-number/5J4VGTBV3DUQP5IZKIIUZLS3IN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5J4VGTBV3DUQP5IZKIIUZLS3IN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5J4VGTBV3DUQP5IZKIIUZLS3IN/action/storage_attestation","attest_author":"https://pith.science/pith/5J4VGTBV3DUQP5IZKIIUZLS3IN/action/author_attestation","sign_citation":"https://pith.science/pith/5J4VGTBV3DUQP5IZKIIUZLS3IN/action/citation_signature","submit_replication":"https://pith.science/pith/5J4VGTBV3DUQP5IZKIIUZLS3IN/action/replication_record"}},"created_at":"2026-07-05T11:12:46.862650+00:00","updated_at":"2026-07-05T11:12:46.862650+00:00"}