{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:6A7UGZZ6QXYKXLIVSXV5T5QPDR","short_pith_number":"pith:6A7UGZZ6","schema_version":"1.0","canonical_sha256":"f03f43673e85f0abad1595ebd9f60f1c7d4b29678d596b410fa58228a42ba901","source":{"kind":"arxiv","id":"2502.04134","version":2},"attestation_state":"computed","paper":{"title":"The Order Effect: Investigating Prompt Sensitivity to Input Order in LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bryan Guan, Mehdi Rezagholizadeh, Peyman Passban, Tanya Roosta","submitted_at":"2025-02-06T15:14:02Z","abstract_excerpt":"As large language models (LLMs) become integral to diverse applications, ensuring their reliability under varying input conditions is crucial. One key issue affecting this reliability is order sensitivity, wherein slight variations in the input arrangement can lead to inconsistent or biased outputs. Although recent advances have reduced this sensitivity, the problem remains unresolved. This paper investigates the extent of order sensitivity in LLMs whose internal components are hidden from users (such as closed-source models or those accessed via API calls). We conduct experiments across multi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.04134","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-02-06T15:14:02Z","cross_cats_sorted":[],"title_canon_sha256":"2b697b392693c37de6a256ebb459bd95a21c58d975c22a93f9636a6b96b71fd9","abstract_canon_sha256":"f7efbb12533d2a0bc2f797d00845354df4137b6c058ae4687086af8f15168279"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:00:42.004329Z","signature_b64":"tXcJ40ly+F7zkwyoYRpumi/wsCkwp1Vq2b0AMTuKI7lJtuulLTV8l3wmtWKqG+LPbKkFhKP1nlodbrlsV/9tDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f03f43673e85f0abad1595ebd9f60f1c7d4b29678d596b410fa58228a42ba901","last_reissued_at":"2026-07-05T11:00:42.003846Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:00:42.003846Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Order Effect: Investigating Prompt Sensitivity to Input Order in LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bryan Guan, Mehdi Rezagholizadeh, Peyman Passban, Tanya Roosta","submitted_at":"2025-02-06T15:14:02Z","abstract_excerpt":"As large language models (LLMs) become integral to diverse applications, ensuring their reliability under varying input conditions is crucial. One key issue affecting this reliability is order sensitivity, wherein slight variations in the input arrangement can lead to inconsistent or biased outputs. Although recent advances have reduced this sensitivity, the problem remains unresolved. This paper investigates the extent of order sensitivity in LLMs whose internal components are hidden from users (such as closed-source models or those accessed via API calls). We conduct experiments across multi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.04134","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.04134/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.04134","created_at":"2026-07-05T11:00:42.003901+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.04134v2","created_at":"2026-07-05T11:00:42.003901+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.04134","created_at":"2026-07-05T11:00:42.003901+00:00"},{"alias_kind":"pith_short_12","alias_value":"6A7UGZZ6QXYK","created_at":"2026-07-05T11:00:42.003901+00:00"},{"alias_kind":"pith_short_16","alias_value":"6A7UGZZ6QXYKXLIV","created_at":"2026-07-05T11:00:42.003901+00:00"},{"alias_kind":"pith_short_8","alias_value":"6A7UGZZ6","created_at":"2026-07-05T11:00:42.003901+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07237","citing_title":"When Large Language Models Fail in Healthcare: Evaluating Sensitivity to Prompt Variations","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2603.28561","citing_title":"Fine-Tuning Large Language Models for Cooperative Tactical Deconfliction of Small Unmanned Aerial Systems","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24700","citing_title":"Green Shielding: A User-Centric Approach Towards Trustworthy AI","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22002","citing_title":"When Cow Urine Cures Constipation on YouTube: Limits of LLMs in Detecting Culture-specific Health Misinformation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18512","citing_title":"S2H-DPO: Hardness-Aware Preference Optimization for Vision-Language Models","ref_index":104,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6A7UGZZ6QXYKXLIVSXV5T5QPDR","json":"https://pith.science/pith/6A7UGZZ6QXYKXLIVSXV5T5QPDR.json","graph_json":"https://pith.science/api/pith-number/6A7UGZZ6QXYKXLIVSXV5T5QPDR/graph.json","events_json":"https://pith.science/api/pith-number/6A7UGZZ6QXYKXLIVSXV5T5QPDR/events.json","paper":"https://pith.science/paper/6A7UGZZ6"},"agent_actions":{"view_html":"https://pith.science/pith/6A7UGZZ6QXYKXLIVSXV5T5QPDR","download_json":"https://pith.science/pith/6A7UGZZ6QXYKXLIVSXV5T5QPDR.json","view_paper":"https://pith.science/paper/6A7UGZZ6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.04134&json=true","fetch_graph":"https://pith.science/api/pith-number/6A7UGZZ6QXYKXLIVSXV5T5QPDR/graph.json","fetch_events":"https://pith.science/api/pith-number/6A7UGZZ6QXYKXLIVSXV5T5QPDR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6A7UGZZ6QXYKXLIVSXV5T5QPDR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6A7UGZZ6QXYKXLIVSXV5T5QPDR/action/storage_attestation","attest_author":"https://pith.science/pith/6A7UGZZ6QXYKXLIVSXV5T5QPDR/action/author_attestation","sign_citation":"https://pith.science/pith/6A7UGZZ6QXYKXLIVSXV5T5QPDR/action/citation_signature","submit_replication":"https://pith.science/pith/6A7UGZZ6QXYKXLIVSXV5T5QPDR/action/replication_record"}},"created_at":"2026-07-05T11:00:42.003901+00:00","updated_at":"2026-07-05T11:00:42.003901+00:00"}