{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RC5PKCBUDOSUMMVYH63GSZWR2W","short_pith_number":"pith:RC5PKCBU","schema_version":"1.0","canonical_sha256":"88baf508341ba54632b83fb66966d1d59f0c031df03fe96f20e6ba76a687e9e6","source":{"kind":"arxiv","id":"2410.12405","version":1},"attestation_state":"computed","paper":{"title":"ProSA: Assessing and Understanding the Prompt Sensitivity of LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dahua Lin, Haodong Duan, Jingming Zhuo, Kai Chen, Songyang Zhang, Xinyu Fang","submitted_at":"2024-10-16T09:38:13Z","abstract_excerpt":"Large language models (LLMs) have demonstrated impressive capabilities across various tasks, but their performance is highly sensitive to the prompts utilized. This variability poses challenges for accurate assessment and user satisfaction. Current research frequently overlooks instance-level prompt variations and their implications on subjective evaluations. To address these shortcomings, we introduce ProSA, a framework designed to evaluate and comprehend prompt sensitivity in LLMs. ProSA incorporates a novel sensitivity metric, PromptSensiScore, and leverages decoding confidence to elucidate"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.12405","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-16T09:38:13Z","cross_cats_sorted":[],"title_canon_sha256":"735e500f627580a8eaefb60ce3ea847741e313aacac78c361274d627b0e1368c","abstract_canon_sha256":"c0abbe4ae4aaa26ee3809421a9141fe995a355bcc65e049017f809f7e4e95230"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:21:18.465849Z","signature_b64":"kcGp/7z2DIBIfO7cXkJcT7z//sWuBQ5KwDMnKSE003NvttdPLtJ/Hq+++35SyNMlL5vygVgDhmv1s7BvQK7mAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"88baf508341ba54632b83fb66966d1d59f0c031df03fe96f20e6ba76a687e9e6","last_reissued_at":"2026-07-05T09:21:18.465376Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:21:18.465376Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ProSA: Assessing and Understanding the Prompt Sensitivity of LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dahua Lin, Haodong Duan, Jingming Zhuo, Kai Chen, Songyang Zhang, Xinyu Fang","submitted_at":"2024-10-16T09:38:13Z","abstract_excerpt":"Large language models (LLMs) have demonstrated impressive capabilities across various tasks, but their performance is highly sensitive to the prompts utilized. This variability poses challenges for accurate assessment and user satisfaction. Current research frequently overlooks instance-level prompt variations and their implications on subjective evaluations. To address these shortcomings, we introduce ProSA, a framework designed to evaluate and comprehend prompt sensitivity in LLMs. ProSA incorporates a novel sensitivity metric, PromptSensiScore, and leverages decoding confidence to elucidate"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.12405","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.12405/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.12405","created_at":"2026-07-05T09:21:18.465434+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.12405v1","created_at":"2026-07-05T09:21:18.465434+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.12405","created_at":"2026-07-05T09:21:18.465434+00:00"},{"alias_kind":"pith_short_12","alias_value":"RC5PKCBUDOSU","created_at":"2026-07-05T09:21:18.465434+00:00"},{"alias_kind":"pith_short_16","alias_value":"RC5PKCBUDOSUMMVY","created_at":"2026-07-05T09:21:18.465434+00:00"},{"alias_kind":"pith_short_8","alias_value":"RC5PKCBU","created_at":"2026-07-05T09:21:18.465434+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.19590","citing_title":"Position: AI Evaluations Should be Grounded on a Theory of Capability","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21465","citing_title":"Leveraging LLMs for Grammar Adaptation: A Study on Metamodel-Grammar Co-Evolution","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2603.10477","citing_title":"PEEM: Prompt Engineering Evaluation Metrics for Interpretable Joint Evaluation of Prompts and Responses","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24700","citing_title":"Green Shielding: A User-Centric Approach Towards Trustworthy AI","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17267","citing_title":"Rectification Difficulty and Optimal Sample Allocation in LLM-Augmented Surveys","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RC5PKCBUDOSUMMVYH63GSZWR2W","json":"https://pith.science/pith/RC5PKCBUDOSUMMVYH63GSZWR2W.json","graph_json":"https://pith.science/api/pith-number/RC5PKCBUDOSUMMVYH63GSZWR2W/graph.json","events_json":"https://pith.science/api/pith-number/RC5PKCBUDOSUMMVYH63GSZWR2W/events.json","paper":"https://pith.science/paper/RC5PKCBU"},"agent_actions":{"view_html":"https://pith.science/pith/RC5PKCBUDOSUMMVYH63GSZWR2W","download_json":"https://pith.science/pith/RC5PKCBUDOSUMMVYH63GSZWR2W.json","view_paper":"https://pith.science/paper/RC5PKCBU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.12405&json=true","fetch_graph":"https://pith.science/api/pith-number/RC5PKCBUDOSUMMVYH63GSZWR2W/graph.json","fetch_events":"https://pith.science/api/pith-number/RC5PKCBUDOSUMMVYH63GSZWR2W/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RC5PKCBUDOSUMMVYH63GSZWR2W/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RC5PKCBUDOSUMMVYH63GSZWR2W/action/storage_attestation","attest_author":"https://pith.science/pith/RC5PKCBUDOSUMMVYH63GSZWR2W/action/author_attestation","sign_citation":"https://pith.science/pith/RC5PKCBUDOSUMMVYH63GSZWR2W/action/citation_signature","submit_replication":"https://pith.science/pith/RC5PKCBUDOSUMMVYH63GSZWR2W/action/replication_record"}},"created_at":"2026-07-05T09:21:18.465434+00:00","updated_at":"2026-07-05T09:21:18.465434+00:00"}