{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:O6EVJ72WJLLLIAWE335XA73NHG","short_pith_number":"pith:O6EVJ72W","schema_version":"1.0","canonical_sha256":"778954ff564ad6b402c4defb707f6d39aa81dea50f4a32763404eebf3af859e4","source":{"kind":"arxiv","id":"2504.06969","version":1},"attestation_state":"computed","paper":{"title":"Towards LLMs Robustness to Changes in Prompt Format Styles","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jason Tsay, Kiran Kate, Lilian Ngweta, Yara Rizk","submitted_at":"2025-04-09T15:26:00Z","abstract_excerpt":"Large language models (LLMs) have gained popularity in recent years for their utility in various applications. However, they are sensitive to non-semantic changes in prompt formats, where small changes in the prompt format can lead to significant performance fluctuations. In the literature, this problem is commonly referred to as prompt brittleness. Previous research on prompt engineering has focused mainly on developing techniques for identifying the optimal prompt for specific tasks. Some studies have also explored the issue of prompt brittleness and proposed methods to quantify performance "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.06969","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-04-09T15:26:00Z","cross_cats_sorted":[],"title_canon_sha256":"36a32f6b1311ff0b3e72e166ad50c47a71810a0597e939b853236be4e5792381","abstract_canon_sha256":"0377c9b2403e0efe7f655c45cd8c8c78ab2e3bb5669b5e5656df4f79a2dae55b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:46:49.251028Z","signature_b64":"qyth0QPHAWHjZm3Uu8EfOcC9ITPVTUL4RbkDx/9RPgmrEEvGBAFjQ1XN+k6WSeJ/oi8odFRxgV36/8+pMUaPBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"778954ff564ad6b402c4defb707f6d39aa81dea50f4a32763404eebf3af859e4","last_reissued_at":"2026-07-05T10:46:49.250530Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:46:49.250530Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards LLMs Robustness to Changes in Prompt Format Styles","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jason Tsay, Kiran Kate, Lilian Ngweta, Yara Rizk","submitted_at":"2025-04-09T15:26:00Z","abstract_excerpt":"Large language models (LLMs) have gained popularity in recent years for their utility in various applications. However, they are sensitive to non-semantic changes in prompt formats, where small changes in the prompt format can lead to significant performance fluctuations. In the literature, this problem is commonly referred to as prompt brittleness. Previous research on prompt engineering has focused mainly on developing techniques for identifying the optimal prompt for specific tasks. Some studies have also explored the issue of prompt brittleness and proposed methods to quantify performance "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.06969","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.06969/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.06969","created_at":"2026-07-05T10:46:49.250596+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.06969v1","created_at":"2026-07-05T10:46:49.250596+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.06969","created_at":"2026-07-05T10:46:49.250596+00:00"},{"alias_kind":"pith_short_12","alias_value":"O6EVJ72WJLLL","created_at":"2026-07-05T10:46:49.250596+00:00"},{"alias_kind":"pith_short_16","alias_value":"O6EVJ72WJLLLIAWE","created_at":"2026-07-05T10:46:49.250596+00:00"},{"alias_kind":"pith_short_8","alias_value":"O6EVJ72W","created_at":"2026-07-05T10:46:49.250596+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.24700","citing_title":"Green Shielding: A User-Centric Approach Towards Trustworthy AI","ref_index":41,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O6EVJ72WJLLLIAWE335XA73NHG","json":"https://pith.science/pith/O6EVJ72WJLLLIAWE335XA73NHG.json","graph_json":"https://pith.science/api/pith-number/O6EVJ72WJLLLIAWE335XA73NHG/graph.json","events_json":"https://pith.science/api/pith-number/O6EVJ72WJLLLIAWE335XA73NHG/events.json","paper":"https://pith.science/paper/O6EVJ72W"},"agent_actions":{"view_html":"https://pith.science/pith/O6EVJ72WJLLLIAWE335XA73NHG","download_json":"https://pith.science/pith/O6EVJ72WJLLLIAWE335XA73NHG.json","view_paper":"https://pith.science/paper/O6EVJ72W","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.06969&json=true","fetch_graph":"https://pith.science/api/pith-number/O6EVJ72WJLLLIAWE335XA73NHG/graph.json","fetch_events":"https://pith.science/api/pith-number/O6EVJ72WJLLLIAWE335XA73NHG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O6EVJ72WJLLLIAWE335XA73NHG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O6EVJ72WJLLLIAWE335XA73NHG/action/storage_attestation","attest_author":"https://pith.science/pith/O6EVJ72WJLLLIAWE335XA73NHG/action/author_attestation","sign_citation":"https://pith.science/pith/O6EVJ72WJLLLIAWE335XA73NHG/action/citation_signature","submit_replication":"https://pith.science/pith/O6EVJ72WJLLLIAWE335XA73NHG/action/replication_record"}},"created_at":"2026-07-05T10:46:49.250596+00:00","updated_at":"2026-07-05T10:46:49.250596+00:00"}