{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:W75M2BNHNH37IDUSETOTF2EI2E","short_pith_number":"pith:W75M2BNH","schema_version":"1.0","canonical_sha256":"b7facd05a769f7f40e9224dd32e888d10734d043eff124daea828273f4859999","source":{"kind":"arxiv","id":"2402.10962","version":4},"attestation_state":"computed","paper":{"title":"Measuring and Controlling Instruction (In)Stability in Language Model Dialogs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"David Bau, Fernanda Vi\\'egas, Hanspeter Pfister, Kenneth Li, Martin Wattenberg, Naomi Bashkansky, Tianle Liu","submitted_at":"2024-02-13T20:10:29Z","abstract_excerpt":"System-prompting is a standard tool for customizing language-model chatbots, enabling them to follow a specific instruction. An implicit assumption in the use of system prompts is that they will be stable, so the chatbot will continue to generate text according to the stipulated instructions for the duration of a conversation. We propose a quantitative benchmark to test this assumption, evaluating instruction stability via self-chats between two instructed chatbots. Testing popular models like LLaMA2-chat-70B and GPT-3.5, we reveal a significant instruction drift within eight rounds of convers"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.10962","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-02-13T20:10:29Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"6b455fc789dd514de84eb7b4b4fcf33611fbae3a4956c1fb725d0a09cdfee312","abstract_canon_sha256":"379b994dede12a62392ecdaa4c1dbe977f108a96997d12b70c17615248957ca3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:48:41.403949Z","signature_b64":"LhBydJGse2QyNqjgnByqy3iGoq+dfZvNR75nTPwJi+j/k0e17DMGTGY3XuMmdqvWJKe6zr1uUDcYi9aqUKB3Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b7facd05a769f7f40e9224dd32e888d10734d043eff124daea828273f4859999","last_reissued_at":"2026-07-05T08:48:41.403480Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:48:41.403480Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Measuring and Controlling Instruction (In)Stability in Language Model Dialogs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"David Bau, Fernanda Vi\\'egas, Hanspeter Pfister, Kenneth Li, Martin Wattenberg, Naomi Bashkansky, Tianle Liu","submitted_at":"2024-02-13T20:10:29Z","abstract_excerpt":"System-prompting is a standard tool for customizing language-model chatbots, enabling them to follow a specific instruction. An implicit assumption in the use of system prompts is that they will be stable, so the chatbot will continue to generate text according to the stipulated instructions for the duration of a conversation. We propose a quantitative benchmark to test this assumption, evaluating instruction stability via self-chats between two instructed chatbots. Testing popular models like LLaMA2-chat-70B and GPT-3.5, we reveal a significant instruction drift within eight rounds of convers"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.10962","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.10962/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.10962","created_at":"2026-07-05T08:48:41.403537+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.10962v4","created_at":"2026-07-05T08:48:41.403537+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.10962","created_at":"2026-07-05T08:48:41.403537+00:00"},{"alias_kind":"pith_short_12","alias_value":"W75M2BNHNH37","created_at":"2026-07-05T08:48:41.403537+00:00"},{"alias_kind":"pith_short_16","alias_value":"W75M2BNHNH37IDUS","created_at":"2026-07-05T08:48:41.403537+00:00"},{"alias_kind":"pith_short_8","alias_value":"W75M2BNH","created_at":"2026-07-05T08:48:41.403537+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07918","citing_title":"Efficient Safety Alignment of Language Models via Latent Personality Traits","ref_index":21,"is_internal_anchor":true},{"citing_arxiv_id":"2606.21843","citing_title":"Measuring What Persists: Conditioning Mechanisms and a Geometric Framework for AI Agent Identity","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07539","citing_title":"Prompt Governance? On Governing Technologies Governed by Natural Language","ref_index":190,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24279","citing_title":"ContextEcho: A Benchmark for Persona Drift in Long Agentic-Coding Sessions","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30981","citing_title":"Cognitive Fatigue in Autoregressive Transformers: Formalization and Measurement","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18226","citing_title":"Context Memorization for Efficient Long Context Generation","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15455","citing_title":"Multi-Turn Neural Transparency: Surfacing Neural Activations Improves User Calibration to LLM Behavioral Drift","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06196","citing_title":"The Granularity Axis: A Micro-to-Macro Latent Direction for Social Roles in Language Models","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09212","citing_title":"SPASM: Stable Persona-driven Agent Simulation for Multi-turn Dialogue Generation","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/W75M2BNHNH37IDUSETOTF2EI2E","json":"https://pith.science/pith/W75M2BNHNH37IDUSETOTF2EI2E.json","graph_json":"https://pith.science/api/pith-number/W75M2BNHNH37IDUSETOTF2EI2E/graph.json","events_json":"https://pith.science/api/pith-number/W75M2BNHNH37IDUSETOTF2EI2E/events.json","paper":"https://pith.science/paper/W75M2BNH"},"agent_actions":{"view_html":"https://pith.science/pith/W75M2BNHNH37IDUSETOTF2EI2E","download_json":"https://pith.science/pith/W75M2BNHNH37IDUSETOTF2EI2E.json","view_paper":"https://pith.science/paper/W75M2BNH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.10962&json=true","fetch_graph":"https://pith.science/api/pith-number/W75M2BNHNH37IDUSETOTF2EI2E/graph.json","fetch_events":"https://pith.science/api/pith-number/W75M2BNHNH37IDUSETOTF2EI2E/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/W75M2BNHNH37IDUSETOTF2EI2E/action/timestamp_anchor","attest_storage":"https://pith.science/pith/W75M2BNHNH37IDUSETOTF2EI2E/action/storage_attestation","attest_author":"https://pith.science/pith/W75M2BNHNH37IDUSETOTF2EI2E/action/author_attestation","sign_citation":"https://pith.science/pith/W75M2BNHNH37IDUSETOTF2EI2E/action/citation_signature","submit_replication":"https://pith.science/pith/W75M2BNHNH37IDUSETOTF2EI2E/action/replication_record"}},"created_at":"2026-07-05T08:48:41.403537+00:00","updated_at":"2026-07-05T08:48:41.403537+00:00"}