{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:6KJWE2K6N7PYQSHWAWZFYS7727","short_pith_number":"pith:6KJWE2K6","schema_version":"1.0","canonical_sha256":"f29362695e6fdf8848f605b25c4bffd7eed2afff67a9b9ff11a0bfd67f74e10c","source":{"kind":"arxiv","id":"2501.19085","version":1},"attestation_state":"computed","paper":{"title":"Enhancing Code Generation for Low-Resource Languages: No Silver Bullet","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Alberto Martin-Lopez, Alessandro Giagnorio, Gabriele Bavota","submitted_at":"2025-01-31T12:23:28Z","abstract_excerpt":"The advent of Large Language Models (LLMs) has significantly advanced the field of automated code generation. LLMs rely on large and diverse datasets to learn syntax, semantics, and usage patterns of programming languages. For low-resource languages (i.e., niche programming languages characterized by the scarcity of training data), the limited availability of such data hampers the models' ability to generalize effectively, resulting in poorer code generation performance as compared to high-resource languages. For this reason, there is a quest for techniques able to close this performance gap. "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.19085","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.SE","submitted_at":"2025-01-31T12:23:28Z","cross_cats_sorted":[],"title_canon_sha256":"6bf44f8ed3456e3ed358ebbf52f0b591b3a0fe2fd76340ce8506ca62ab36544c","abstract_canon_sha256":"b787ad3714cf4d203ab313cc7c20a74c04f71825e89c1889a4c5d8050a8d3d37"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:07:55.906366Z","signature_b64":"/x6td5svMIZsAqYg/naiNmkuh7aEyOnKh04cdhpgiGDGgr3P3XygtJxJq3f6q0zpdAyQ3cqxQm7NqIqEhaRKBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f29362695e6fdf8848f605b25c4bffd7eed2afff67a9b9ff11a0bfd67f74e10c","last_reissued_at":"2026-07-05T10:07:55.905829Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:07:55.905829Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Enhancing Code Generation for Low-Resource Languages: No Silver Bullet","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Alberto Martin-Lopez, Alessandro Giagnorio, Gabriele Bavota","submitted_at":"2025-01-31T12:23:28Z","abstract_excerpt":"The advent of Large Language Models (LLMs) has significantly advanced the field of automated code generation. LLMs rely on large and diverse datasets to learn syntax, semantics, and usage patterns of programming languages. For low-resource languages (i.e., niche programming languages characterized by the scarcity of training data), the limited availability of such data hampers the models' ability to generalize effectively, resulting in poorer code generation performance as compared to high-resource languages. For this reason, there is a quest for techniques able to close this performance gap. "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.19085","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.19085/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.19085","created_at":"2026-07-05T10:07:55.905888+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.19085v1","created_at":"2026-07-05T10:07:55.905888+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.19085","created_at":"2026-07-05T10:07:55.905888+00:00"},{"alias_kind":"pith_short_12","alias_value":"6KJWE2K6N7PY","created_at":"2026-07-05T10:07:55.905888+00:00"},{"alias_kind":"pith_short_16","alias_value":"6KJWE2K6N7PYQSHW","created_at":"2026-07-05T10:07:55.905888+00:00"},{"alias_kind":"pith_short_8","alias_value":"6KJWE2K6","created_at":"2026-07-05T10:07:55.905888+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07748","citing_title":"Selective Left-Shift: Turning Test-Time Compute and Difficulty-based Curation into Training Data for Low-Resource Code Generation","ref_index":9,"is_internal_anchor":true},{"citing_arxiv_id":"2512.00380","citing_title":"Knowledge-Graph-Driven Data Synthesis for Low-Resource Software Development: A HarmonyOS Case Study","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25960","citing_title":"Large Language Models for Multilingual Code Intelligence: A Survey","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17010","citing_title":"Improving LLM Code Reasoning via Semantic Equivalence Self-Play with Formal Verification","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6KJWE2K6N7PYQSHWAWZFYS7727","json":"https://pith.science/pith/6KJWE2K6N7PYQSHWAWZFYS7727.json","graph_json":"https://pith.science/api/pith-number/6KJWE2K6N7PYQSHWAWZFYS7727/graph.json","events_json":"https://pith.science/api/pith-number/6KJWE2K6N7PYQSHWAWZFYS7727/events.json","paper":"https://pith.science/paper/6KJWE2K6"},"agent_actions":{"view_html":"https://pith.science/pith/6KJWE2K6N7PYQSHWAWZFYS7727","download_json":"https://pith.science/pith/6KJWE2K6N7PYQSHWAWZFYS7727.json","view_paper":"https://pith.science/paper/6KJWE2K6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.19085&json=true","fetch_graph":"https://pith.science/api/pith-number/6KJWE2K6N7PYQSHWAWZFYS7727/graph.json","fetch_events":"https://pith.science/api/pith-number/6KJWE2K6N7PYQSHWAWZFYS7727/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6KJWE2K6N7PYQSHWAWZFYS7727/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6KJWE2K6N7PYQSHWAWZFYS7727/action/storage_attestation","attest_author":"https://pith.science/pith/6KJWE2K6N7PYQSHWAWZFYS7727/action/author_attestation","sign_citation":"https://pith.science/pith/6KJWE2K6N7PYQSHWAWZFYS7727/action/citation_signature","submit_replication":"https://pith.science/pith/6KJWE2K6N7PYQSHWAWZFYS7727/action/replication_record"}},"created_at":"2026-07-05T10:07:55.905888+00:00","updated_at":"2026-07-05T10:07:55.905888+00:00"}