{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:F2MGEQLVDW4XXVWFWMOKDZZ3RR","short_pith_number":"pith:F2MGEQLV","schema_version":"1.0","canonical_sha256":"2e986241751db97bd6c5b31ca1e73b8c447cf6c884dd5896a9d5c95a9e2b9b3a","source":{"kind":"arxiv","id":"2410.02185","version":2},"attestation_state":"computed","paper":{"title":"POSIX: A Prompt Sensitivity Index For Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Anwoy Chatterjee, H S V N S Kowndinya Renduchintala, Sumit Bhatia, Tanmoy Chakraborty","submitted_at":"2024-10-03T04:01:14Z","abstract_excerpt":"Despite their remarkable capabilities, Large Language Models (LLMs) are found to be surprisingly sensitive to minor variations in prompts, often generating significantly divergent outputs in response to minor variations in the prompts, such as spelling errors, alteration of wording or the prompt template. However, while assessing the quality of an LLM, the focus often tends to be solely on its performance on downstream tasks, while very little to no attention is paid to prompt sensitivity. To fill this gap, we propose POSIX - a novel PrOmpt Sensitivity IndeX as a reliable measure of prompt sen"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.02185","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-03T04:01:14Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"d3a5eabcd85bbba7a35a301d3f63a9d8cfef5a80e135e86943cd22aeed8e8013","abstract_canon_sha256":"6a1343997e010d945cc21ad6eb6f4f5b12fa20ed15e97f1da1f69d7845f1221d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:15:38.873151Z","signature_b64":"p6dzpHC1+hIpDizCc09PimK1ZgmQtl01DfdcwHnGVKT/q6DhJnc1h7ruPQhbXOk/l0v5ubJktBeN5ovpCqKtAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2e986241751db97bd6c5b31ca1e73b8c447cf6c884dd5896a9d5c95a9e2b9b3a","last_reissued_at":"2026-07-05T09:15:38.872620Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:15:38.872620Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"POSIX: A Prompt Sensitivity Index For Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Anwoy Chatterjee, H S V N S Kowndinya Renduchintala, Sumit Bhatia, Tanmoy Chakraborty","submitted_at":"2024-10-03T04:01:14Z","abstract_excerpt":"Despite their remarkable capabilities, Large Language Models (LLMs) are found to be surprisingly sensitive to minor variations in prompts, often generating significantly divergent outputs in response to minor variations in the prompts, such as spelling errors, alteration of wording or the prompt template. However, while assessing the quality of an LLM, the focus often tends to be solely on its performance on downstream tasks, while very little to no attention is paid to prompt sensitivity. To fill this gap, we propose POSIX - a novel PrOmpt Sensitivity IndeX as a reliable measure of prompt sen"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.02185","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.02185/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.02185","created_at":"2026-07-05T09:15:38.872687+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.02185v2","created_at":"2026-07-05T09:15:38.872687+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.02185","created_at":"2026-07-05T09:15:38.872687+00:00"},{"alias_kind":"pith_short_12","alias_value":"F2MGEQLVDW4X","created_at":"2026-07-05T09:15:38.872687+00:00"},{"alias_kind":"pith_short_16","alias_value":"F2MGEQLVDW4XXVWF","created_at":"2026-07-05T09:15:38.872687+00:00"},{"alias_kind":"pith_short_8","alias_value":"F2MGEQLV","created_at":"2026-07-05T09:15:38.872687+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00937","citing_title":"Persona Non Grata: LLM Persona-Driven Generations in MCQA are Unstable in Distinct Dimensions","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27440","citing_title":"Paraphrase Brittleness in Production Retrieval-Augmented Commercial Recommendation: Reproducibility Below the Rerun-Stability Baseline","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30207","citing_title":"Persona Conditioning of Brand Recommendations in Retrieval-Augmented Commercial Chat: A Prominence-Stratified Cross-Provider Audit","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22797","citing_title":"Measuring Behavior Portability in Large Language Models","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2603.10477","citing_title":"PEEM: Prompt Engineering Evaluation Metrics for Interpretable Joint Evaluation of Prompts and Responses","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F2MGEQLVDW4XXVWFWMOKDZZ3RR","json":"https://pith.science/pith/F2MGEQLVDW4XXVWFWMOKDZZ3RR.json","graph_json":"https://pith.science/api/pith-number/F2MGEQLVDW4XXVWFWMOKDZZ3RR/graph.json","events_json":"https://pith.science/api/pith-number/F2MGEQLVDW4XXVWFWMOKDZZ3RR/events.json","paper":"https://pith.science/paper/F2MGEQLV"},"agent_actions":{"view_html":"https://pith.science/pith/F2MGEQLVDW4XXVWFWMOKDZZ3RR","download_json":"https://pith.science/pith/F2MGEQLVDW4XXVWFWMOKDZZ3RR.json","view_paper":"https://pith.science/paper/F2MGEQLV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.02185&json=true","fetch_graph":"https://pith.science/api/pith-number/F2MGEQLVDW4XXVWFWMOKDZZ3RR/graph.json","fetch_events":"https://pith.science/api/pith-number/F2MGEQLVDW4XXVWFWMOKDZZ3RR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F2MGEQLVDW4XXVWFWMOKDZZ3RR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F2MGEQLVDW4XXVWFWMOKDZZ3RR/action/storage_attestation","attest_author":"https://pith.science/pith/F2MGEQLVDW4XXVWFWMOKDZZ3RR/action/author_attestation","sign_citation":"https://pith.science/pith/F2MGEQLVDW4XXVWFWMOKDZZ3RR/action/citation_signature","submit_replication":"https://pith.science/pith/F2MGEQLVDW4XXVWFWMOKDZZ3RR/action/replication_record"}},"created_at":"2026-07-05T09:15:38.872687+00:00","updated_at":"2026-07-05T09:15:38.872687+00:00"}