{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:SU7WKYNURCBFMCSTP5DCH3H6UJ","short_pith_number":"pith:SU7WKYNU","schema_version":"1.0","canonical_sha256":"953f6561b48882560a537f4623ecfea26a831fde0f02e5e7c4f8fdafd7f43114","source":{"kind":"arxiv","id":"2607.20454","version":1},"attestation_state":"computed","paper":{"title":"Response drift across frontier large language models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Ali Aledhari, Fatimah Aledhari, Gowtham Venkat Eathamokkala, Mohamed Rahouti, Mohammed Aledhari","submitted_at":"2026-05-14T21:06:34Z","abstract_excerpt":"All frontier large language models (LLMs) exhibit response drift -- producing outputs that deviate from expert-validated references -- yet the magnitude and structure of this drift remain uncharacterised by systematic human evaluation. Here we report a fully crossed evaluation in which 47 geographically diverse participants each assessed all 62 multidomain questions across ten frontier LLMs under blinded conditions, yielding 29,140 independent assessments. Every model drifts, but drift magnitude varies substantially: eight models converge on a statistically indistinguishable ceiling (78-81% de"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.20454","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-05-14T21:06:34Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a67f26e8d998462904b549928fe706e3d7f1f2f078fbe2ed2d9d65f25c2acd70","abstract_canon_sha256":"cc776796e8446e8f6e385276d0c29d55209b7ace10aa34b2ee6bd13100b75d58"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-24T00:23:16.334478Z","signature_b64":"YCHbtycLrv2qlZJkPZvjH9GD5h9IGHHf5utkL6oFlNqxqtYgQ/75muErvXYHUKaYO5NPwqekOusEH08zF/8OBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"953f6561b48882560a537f4623ecfea26a831fde0f02e5e7c4f8fdafd7f43114","last_reissued_at":"2026-07-24T00:23:16.333580Z","signature_status":"signed_v1","first_computed_at":"2026-07-24T00:23:16.333580Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Response drift across frontier large language models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Ali Aledhari, Fatimah Aledhari, Gowtham Venkat Eathamokkala, Mohamed Rahouti, Mohammed Aledhari","submitted_at":"2026-05-14T21:06:34Z","abstract_excerpt":"All frontier large language models (LLMs) exhibit response drift -- producing outputs that deviate from expert-validated references -- yet the magnitude and structure of this drift remain uncharacterised by systematic human evaluation. Here we report a fully crossed evaluation in which 47 geographically diverse participants each assessed all 62 multidomain questions across ten frontier LLMs under blinded conditions, yielding 29,140 independent assessments. Every model drifts, but drift magnitude varies substantially: eight models converge on a statistically indistinguishable ceiling (78-81% de"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.20454","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.20454/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.20454","created_at":"2026-07-24T00:23:16.334040+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.20454v1","created_at":"2026-07-24T00:23:16.334040+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.20454","created_at":"2026-07-24T00:23:16.334040+00:00"},{"alias_kind":"pith_short_12","alias_value":"SU7WKYNURCBF","created_at":"2026-07-24T00:23:16.334040+00:00"},{"alias_kind":"pith_short_16","alias_value":"SU7WKYNURCBFMCST","created_at":"2026-07-24T00:23:16.334040+00:00"},{"alias_kind":"pith_short_8","alias_value":"SU7WKYNU","created_at":"2026-07-24T00:23:16.334040+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SU7WKYNURCBFMCSTP5DCH3H6UJ","json":"https://pith.science/pith/SU7WKYNURCBFMCSTP5DCH3H6UJ.json","graph_json":"https://pith.science/api/pith-number/SU7WKYNURCBFMCSTP5DCH3H6UJ/graph.json","events_json":"https://pith.science/api/pith-number/SU7WKYNURCBFMCSTP5DCH3H6UJ/events.json","paper":"https://pith.science/paper/SU7WKYNU"},"agent_actions":{"view_html":"https://pith.science/pith/SU7WKYNURCBFMCSTP5DCH3H6UJ","download_json":"https://pith.science/pith/SU7WKYNURCBFMCSTP5DCH3H6UJ.json","view_paper":"https://pith.science/paper/SU7WKYNU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.20454&json=true","fetch_graph":"https://pith.science/api/pith-number/SU7WKYNURCBFMCSTP5DCH3H6UJ/graph.json","fetch_events":"https://pith.science/api/pith-number/SU7WKYNURCBFMCSTP5DCH3H6UJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SU7WKYNURCBFMCSTP5DCH3H6UJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SU7WKYNURCBFMCSTP5DCH3H6UJ/action/storage_attestation","attest_author":"https://pith.science/pith/SU7WKYNURCBFMCSTP5DCH3H6UJ/action/author_attestation","sign_citation":"https://pith.science/pith/SU7WKYNURCBFMCSTP5DCH3H6UJ/action/citation_signature","submit_replication":"https://pith.science/pith/SU7WKYNURCBFMCSTP5DCH3H6UJ/action/replication_record"}},"created_at":"2026-07-24T00:23:16.334040+00:00","updated_at":"2026-07-24T00:23:16.334040+00:00"}