{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3D7SEQW6M6NPHWLSIVOKVUPPFB","short_pith_number":"pith:3D7SEQW6","schema_version":"1.0","canonical_sha256":"d8ff2242de679af3d972455caad1ef287800d287730ef1845da208cdf6b0cdd9","source":{"kind":"arxiv","id":"2411.19508","version":1},"attestation_state":"computed","paper":{"title":"On the Adversarial Robustness of Instruction-Tuned Large Language Models for Code","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CR"],"primary_cat":"cs.SE","authors_text":"Md Imran Hossen, Xiali Hei","submitted_at":"2024-11-29T07:00:47Z","abstract_excerpt":"The advent of instruction-tuned Large Language Models designed for coding tasks (Code LLMs) has transformed software engineering practices. However, their robustness against various input challenges remains a critical concern. This study introduces DegradePrompter, a novel method designed to systematically evaluate the robustness of instruction-tuned Code LLMs. We assess the impact of diverse input challenges on the functionality and correctness of generated code using rigorous metrics and established benchmarks. Our comprehensive evaluation includes five state-of-the-art open-source models an"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.19508","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2024-11-29T07:00:47Z","cross_cats_sorted":["cs.CR"],"title_canon_sha256":"73799c45e5ba033d6c3b79c0fe253b1e6304fd87f29c654c7661db2336b7007f","abstract_canon_sha256":"4432be331d27aefbd200da5d81d1712ca3eca996e3d52ff4e95c34d3d0c24454"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:42:03.426941Z","signature_b64":"GH25gexf2CG/mUGNVnkcV6lCRUTK7nFh2kK5L5FKcA+FDcM6Jl02zAe1MFpu5NOuPUXy8NQbWuTUu/X3Y+2TDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d8ff2242de679af3d972455caad1ef287800d287730ef1845da208cdf6b0cdd9","last_reissued_at":"2026-07-05T09:42:03.426447Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:42:03.426447Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On the Adversarial Robustness of Instruction-Tuned Large Language Models for Code","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CR"],"primary_cat":"cs.SE","authors_text":"Md Imran Hossen, Xiali Hei","submitted_at":"2024-11-29T07:00:47Z","abstract_excerpt":"The advent of instruction-tuned Large Language Models designed for coding tasks (Code LLMs) has transformed software engineering practices. However, their robustness against various input challenges remains a critical concern. This study introduces DegradePrompter, a novel method designed to systematically evaluate the robustness of instruction-tuned Code LLMs. We assess the impact of diverse input challenges on the functionality and correctness of generated code using rigorous metrics and established benchmarks. Our comprehensive evaluation includes five state-of-the-art open-source models an"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.19508","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.19508/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.19508","created_at":"2026-07-05T09:42:03.426512+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.19508v1","created_at":"2026-07-05T09:42:03.426512+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.19508","created_at":"2026-07-05T09:42:03.426512+00:00"},{"alias_kind":"pith_short_12","alias_value":"3D7SEQW6M6NP","created_at":"2026-07-05T09:42:03.426512+00:00"},{"alias_kind":"pith_short_16","alias_value":"3D7SEQW6M6NPHWLS","created_at":"2026-07-05T09:42:03.426512+00:00"},{"alias_kind":"pith_short_8","alias_value":"3D7SEQW6","created_at":"2026-07-05T09:42:03.426512+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03606","citing_title":"Testing LLM Arithmetic Reasoning Generalization with Automatic Numeric-Remapping Attacks","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3D7SEQW6M6NPHWLSIVOKVUPPFB","json":"https://pith.science/pith/3D7SEQW6M6NPHWLSIVOKVUPPFB.json","graph_json":"https://pith.science/api/pith-number/3D7SEQW6M6NPHWLSIVOKVUPPFB/graph.json","events_json":"https://pith.science/api/pith-number/3D7SEQW6M6NPHWLSIVOKVUPPFB/events.json","paper":"https://pith.science/paper/3D7SEQW6"},"agent_actions":{"view_html":"https://pith.science/pith/3D7SEQW6M6NPHWLSIVOKVUPPFB","download_json":"https://pith.science/pith/3D7SEQW6M6NPHWLSIVOKVUPPFB.json","view_paper":"https://pith.science/paper/3D7SEQW6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.19508&json=true","fetch_graph":"https://pith.science/api/pith-number/3D7SEQW6M6NPHWLSIVOKVUPPFB/graph.json","fetch_events":"https://pith.science/api/pith-number/3D7SEQW6M6NPHWLSIVOKVUPPFB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3D7SEQW6M6NPHWLSIVOKVUPPFB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3D7SEQW6M6NPHWLSIVOKVUPPFB/action/storage_attestation","attest_author":"https://pith.science/pith/3D7SEQW6M6NPHWLSIVOKVUPPFB/action/author_attestation","sign_citation":"https://pith.science/pith/3D7SEQW6M6NPHWLSIVOKVUPPFB/action/citation_signature","submit_replication":"https://pith.science/pith/3D7SEQW6M6NPHWLSIVOKVUPPFB/action/replication_record"}},"created_at":"2026-07-05T09:42:03.426512+00:00","updated_at":"2026-07-05T09:42:03.426512+00:00"}