{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:3JROBPOHFX5W5UK33ND2GCLB27","short_pith_number":"pith:3JROBPOH","schema_version":"1.0","canonical_sha256":"da62e0bdc72dfb6ed15bdb47a30961d7edaf59205736f2b59864004ba8b3ed1b","source":{"kind":"arxiv","id":"2502.06065","version":1},"attestation_state":"computed","paper":{"title":"Benchmarking Prompt Sensitivity in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.IR"],"primary_cat":"cs.CL","authors_text":"Amirhossein Razavi, Ebrahim Bagheri, Mina Soltangheis, Morteza Zihayat, Negar Arabzadeh, Sara Salamat","submitted_at":"2025-02-09T23:01:03Z","abstract_excerpt":"Large language Models (LLMs) are highly sensitive to variations in prompt formulation, which can significantly impact their ability to generate accurate responses. In this paper, we introduce a new task, Prompt Sensitivity Prediction, and a dataset PromptSET designed to investigate the effects of slight prompt variations on LLM performance. Using TriviaQA and HotpotQA datasets as the foundation of our work, we generate prompt variations and evaluate their effectiveness across multiple LLMs. We benchmark the prompt sensitivity prediction task employing state-of-the-art methods from related task"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.06065","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-02-09T23:01:03Z","cross_cats_sorted":["cs.AI","cs.IR"],"title_canon_sha256":"2d9043a8f5483593e2e5697fcb2aea599dca87e794ed3c2a08dcab3d332309f6","abstract_canon_sha256":"90a92a0e74803321c6a5e1cd25176bc49b2d74eb34df90673693444e99cc6fcd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:11:47.218940Z","signature_b64":"k7ACr2R5Eepc9Y5wWHOlhGz+ZATFTVdEtbExGPxGUbqr3Jx+PxQkY4SvsK7vBrl0MqV6Yy2QXd4yf2LKONoaCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"da62e0bdc72dfb6ed15bdb47a30961d7edaf59205736f2b59864004ba8b3ed1b","last_reissued_at":"2026-07-05T10:11:47.218379Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:11:47.218379Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Benchmarking Prompt Sensitivity in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.IR"],"primary_cat":"cs.CL","authors_text":"Amirhossein Razavi, Ebrahim Bagheri, Mina Soltangheis, Morteza Zihayat, Negar Arabzadeh, Sara Salamat","submitted_at":"2025-02-09T23:01:03Z","abstract_excerpt":"Large language Models (LLMs) are highly sensitive to variations in prompt formulation, which can significantly impact their ability to generate accurate responses. In this paper, we introduce a new task, Prompt Sensitivity Prediction, and a dataset PromptSET designed to investigate the effects of slight prompt variations on LLM performance. Using TriviaQA and HotpotQA datasets as the foundation of our work, we generate prompt variations and evaluate their effectiveness across multiple LLMs. We benchmark the prompt sensitivity prediction task employing state-of-the-art methods from related task"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.06065","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.06065/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.06065","created_at":"2026-07-05T10:11:47.218477+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.06065v1","created_at":"2026-07-05T10:11:47.218477+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.06065","created_at":"2026-07-05T10:11:47.218477+00:00"},{"alias_kind":"pith_short_12","alias_value":"3JROBPOHFX5W","created_at":"2026-07-05T10:11:47.218477+00:00"},{"alias_kind":"pith_short_16","alias_value":"3JROBPOHFX5W5UK3","created_at":"2026-07-05T10:11:47.218477+00:00"},{"alias_kind":"pith_short_8","alias_value":"3JROBPOH","created_at":"2026-07-05T10:11:47.218477+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.18890","citing_title":"Stop Drawing Scientific Claims from LLM Social Simulations Without Robustness Audits","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08522","citing_title":"Coordinates of Capability: A Unified MTMM-Geometric Framework for LLM Evaluation","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19260","citing_title":"Understanding the Mechanism of Altruism in Large Language Models","ref_index":192,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3JROBPOHFX5W5UK33ND2GCLB27","json":"https://pith.science/pith/3JROBPOHFX5W5UK33ND2GCLB27.json","graph_json":"https://pith.science/api/pith-number/3JROBPOHFX5W5UK33ND2GCLB27/graph.json","events_json":"https://pith.science/api/pith-number/3JROBPOHFX5W5UK33ND2GCLB27/events.json","paper":"https://pith.science/paper/3JROBPOH"},"agent_actions":{"view_html":"https://pith.science/pith/3JROBPOHFX5W5UK33ND2GCLB27","download_json":"https://pith.science/pith/3JROBPOHFX5W5UK33ND2GCLB27.json","view_paper":"https://pith.science/paper/3JROBPOH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.06065&json=true","fetch_graph":"https://pith.science/api/pith-number/3JROBPOHFX5W5UK33ND2GCLB27/graph.json","fetch_events":"https://pith.science/api/pith-number/3JROBPOHFX5W5UK33ND2GCLB27/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3JROBPOHFX5W5UK33ND2GCLB27/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3JROBPOHFX5W5UK33ND2GCLB27/action/storage_attestation","attest_author":"https://pith.science/pith/3JROBPOHFX5W5UK33ND2GCLB27/action/author_attestation","sign_citation":"https://pith.science/pith/3JROBPOHFX5W5UK33ND2GCLB27/action/citation_signature","submit_replication":"https://pith.science/pith/3JROBPOHFX5W5UK33ND2GCLB27/action/replication_record"}},"created_at":"2026-07-05T10:11:47.218477+00:00","updated_at":"2026-07-05T10:11:47.218477+00:00"}