{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OSYBTQOAZUK6KLVHIV2OGALHM3","short_pith_number":"pith:OSYBTQOA","schema_version":"1.0","canonical_sha256":"74b019c1c0cd15e52ea74574e3016766da0b4ece67bca066e707884f2268bbc4","source":{"kind":"arxiv","id":"2406.12334","version":4},"attestation_state":"computed","paper":{"title":"What Did I Do Wrong? Quantifying LLMs' Sensitivity and Consistency to Prompt Engineering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SE"],"primary_cat":"cs.LG","authors_text":"Davide Sanvito, Federico Errica, Giuseppe Siracusano, Roberto Bifulco","submitted_at":"2024-06-18T06:59:24Z","abstract_excerpt":"Large Language Models (LLMs) changed the way we design and interact with software systems. Their ability to process and extract information from text has drastically improved productivity in a number of routine tasks. Developers that want to include these models in their software stack, however, face a dreadful challenge: debugging LLMs' inconsistent behavior across minor variations of the prompt. We therefore introduce two metrics for classification tasks, namely sensitivity and consistency, which are complementary to task performance. First, sensitivity measures changes of predictions across"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.12334","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-06-18T06:59:24Z","cross_cats_sorted":["cs.SE"],"title_canon_sha256":"859e429c5c32a70721a768b3b2ffbdf45b2dc1e4b23002fa0317901fd1df7315","abstract_canon_sha256":"b3d7b7bed1e4d9bc6d344c14b8ba311e6f88a4438d508166ee98bb0f4fbec004"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:57:52.914142Z","signature_b64":"cXZCWym0xEGuy/igYD5AKzowvwGcUlYSEodhO4J15lYUGLz6ItT3/QbUpIoMBD2/9qSb3K1r4O24dwlPHX6QAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"74b019c1c0cd15e52ea74574e3016766da0b4ece67bca066e707884f2268bbc4","last_reissued_at":"2026-07-05T11:57:52.913655Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:57:52.913655Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"What Did I Do Wrong? Quantifying LLMs' Sensitivity and Consistency to Prompt Engineering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SE"],"primary_cat":"cs.LG","authors_text":"Davide Sanvito, Federico Errica, Giuseppe Siracusano, Roberto Bifulco","submitted_at":"2024-06-18T06:59:24Z","abstract_excerpt":"Large Language Models (LLMs) changed the way we design and interact with software systems. Their ability to process and extract information from text has drastically improved productivity in a number of routine tasks. Developers that want to include these models in their software stack, however, face a dreadful challenge: debugging LLMs' inconsistent behavior across minor variations of the prompt. We therefore introduce two metrics for classification tasks, namely sensitivity and consistency, which are complementary to task performance. First, sensitivity measures changes of predictions across"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.12334","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.12334/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.12334","created_at":"2026-07-05T11:57:52.913716+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.12334v4","created_at":"2026-07-05T11:57:52.913716+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.12334","created_at":"2026-07-05T11:57:52.913716+00:00"},{"alias_kind":"pith_short_12","alias_value":"OSYBTQOAZUK6","created_at":"2026-07-05T11:57:52.913716+00:00"},{"alias_kind":"pith_short_16","alias_value":"OSYBTQOAZUK6KLVH","created_at":"2026-07-05T11:57:52.913716+00:00"},{"alias_kind":"pith_short_8","alias_value":"OSYBTQOA","created_at":"2026-07-05T11:57:52.913716+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.19590","citing_title":"Position: AI Evaluations Should be Grounded on a Theory of Capability","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2512.11013","citing_title":"PIAST: Rapid Prompting with In-context Augmentation for Scarce Training data","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2512.12858","citing_title":"Information-Consistent Language Model Recommendations through Group Relative Policy Optimization","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07957","citing_title":"Similar Pattern Annotation via Retrieval Knowledge for LLM-Based Test Code Fault Localization","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OSYBTQOAZUK6KLVHIV2OGALHM3","json":"https://pith.science/pith/OSYBTQOAZUK6KLVHIV2OGALHM3.json","graph_json":"https://pith.science/api/pith-number/OSYBTQOAZUK6KLVHIV2OGALHM3/graph.json","events_json":"https://pith.science/api/pith-number/OSYBTQOAZUK6KLVHIV2OGALHM3/events.json","paper":"https://pith.science/paper/OSYBTQOA"},"agent_actions":{"view_html":"https://pith.science/pith/OSYBTQOAZUK6KLVHIV2OGALHM3","download_json":"https://pith.science/pith/OSYBTQOAZUK6KLVHIV2OGALHM3.json","view_paper":"https://pith.science/paper/OSYBTQOA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.12334&json=true","fetch_graph":"https://pith.science/api/pith-number/OSYBTQOAZUK6KLVHIV2OGALHM3/graph.json","fetch_events":"https://pith.science/api/pith-number/OSYBTQOAZUK6KLVHIV2OGALHM3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OSYBTQOAZUK6KLVHIV2OGALHM3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OSYBTQOAZUK6KLVHIV2OGALHM3/action/storage_attestation","attest_author":"https://pith.science/pith/OSYBTQOAZUK6KLVHIV2OGALHM3/action/author_attestation","sign_citation":"https://pith.science/pith/OSYBTQOAZUK6KLVHIV2OGALHM3/action/citation_signature","submit_replication":"https://pith.science/pith/OSYBTQOAZUK6KLVHIV2OGALHM3/action/replication_record"}},"created_at":"2026-07-05T11:57:52.913716+00:00","updated_at":"2026-07-05T11:57:52.913716+00:00"}