{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:UASZMZX7QPQ72HM6OTVGS24OQE","short_pith_number":"pith:UASZMZX7","schema_version":"1.0","canonical_sha256":"a0259666ff83e1fd1d9e74ea696b8e810c5daccb8d4086d94d9c09062506fab0","source":{"kind":"arxiv","id":"2507.15868","version":1},"attestation_state":"computed","paper":{"title":"Small Edits, Big Consequences: Telling Good from Bad Robustness in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Altynbek Ismailov, Salia Asanova","submitted_at":"2025-07-15T03:22:07Z","abstract_excerpt":"Large language models (LLMs) now write code in settings where misreading a single word can break safety or cost money, yet we still expect them to overlook stray typos. To probe where useful robustness ends and harmful insensitivity begins, we compile 50 LeetCode problems and craft three minimal prompt perturbations that should vary in importance: (i) progressive underspecification deleting 10 % of words per step; (ii) lexical flip swapping a pivotal quantifier (\"max\" to \"min\"); and (iii) jargon inflation replacing a common noun with an obscure technical synonym. Six frontier models, including"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.15868","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-07-15T03:22:07Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"6d72e4b72524d32235aab28a27e56f3070f8099cc8ffb9751961c34565ce5a41","abstract_canon_sha256":"aa378551f2a67bec569bc85e30be60f53fb06c0551364b28c7cd653313bf255f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:40:58.355074Z","signature_b64":"hJhLWYdP7GfnBXgEDZJpXO5t150H7fA4oB8n+vfSCB4mDnjqnfon5jF8J/aPt3VOdldzGqu+206EvV2LV5zgDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a0259666ff83e1fd1d9e74ea696b8e810c5daccb8d4086d94d9c09062506fab0","last_reissued_at":"2026-07-05T11:40:58.354572Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:40:58.354572Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Small Edits, Big Consequences: Telling Good from Bad Robustness in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Altynbek Ismailov, Salia Asanova","submitted_at":"2025-07-15T03:22:07Z","abstract_excerpt":"Large language models (LLMs) now write code in settings where misreading a single word can break safety or cost money, yet we still expect them to overlook stray typos. To probe where useful robustness ends and harmful insensitivity begins, we compile 50 LeetCode problems and craft three minimal prompt perturbations that should vary in importance: (i) progressive underspecification deleting 10 % of words per step; (ii) lexical flip swapping a pivotal quantifier (\"max\" to \"min\"); and (iii) jargon inflation replacing a common noun with an obscure technical synonym. Six frontier models, including"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.15868","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.15868/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.15868","created_at":"2026-07-05T11:40:58.354638+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.15868v1","created_at":"2026-07-05T11:40:58.354638+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.15868","created_at":"2026-07-05T11:40:58.354638+00:00"},{"alias_kind":"pith_short_12","alias_value":"UASZMZX7QPQ7","created_at":"2026-07-05T11:40:58.354638+00:00"},{"alias_kind":"pith_short_16","alias_value":"UASZMZX7QPQ72HM6","created_at":"2026-07-05T11:40:58.354638+00:00"},{"alias_kind":"pith_short_8","alias_value":"UASZMZX7","created_at":"2026-07-05T11:40:58.354638+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.13284","citing_title":"Learning Perturbations to Extrapolate Your LLM","ref_index":60,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UASZMZX7QPQ72HM6OTVGS24OQE","json":"https://pith.science/pith/UASZMZX7QPQ72HM6OTVGS24OQE.json","graph_json":"https://pith.science/api/pith-number/UASZMZX7QPQ72HM6OTVGS24OQE/graph.json","events_json":"https://pith.science/api/pith-number/UASZMZX7QPQ72HM6OTVGS24OQE/events.json","paper":"https://pith.science/paper/UASZMZX7"},"agent_actions":{"view_html":"https://pith.science/pith/UASZMZX7QPQ72HM6OTVGS24OQE","download_json":"https://pith.science/pith/UASZMZX7QPQ72HM6OTVGS24OQE.json","view_paper":"https://pith.science/paper/UASZMZX7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.15868&json=true","fetch_graph":"https://pith.science/api/pith-number/UASZMZX7QPQ72HM6OTVGS24OQE/graph.json","fetch_events":"https://pith.science/api/pith-number/UASZMZX7QPQ72HM6OTVGS24OQE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UASZMZX7QPQ72HM6OTVGS24OQE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UASZMZX7QPQ72HM6OTVGS24OQE/action/storage_attestation","attest_author":"https://pith.science/pith/UASZMZX7QPQ72HM6OTVGS24OQE/action/author_attestation","sign_citation":"https://pith.science/pith/UASZMZX7QPQ72HM6OTVGS24OQE/action/citation_signature","submit_replication":"https://pith.science/pith/UASZMZX7QPQ72HM6OTVGS24OQE/action/replication_record"}},"created_at":"2026-07-05T11:40:58.354638+00:00","updated_at":"2026-07-05T11:40:58.354638+00:00"}