{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:L3WYWM7X4X7PSRXWK2V2K7BUMA","short_pith_number":"pith:L3WYWM7X","schema_version":"1.0","canonical_sha256":"5eed8b33f7e5fef946f656aba57c346037703961397fe9dae99fac70c2cb5ff8","source":{"kind":"arxiv","id":"2608.05004","version":1},"attestation_state":"computed","paper":{"title":"DelusionEval: Measuring Delusion-Linked Behaviors in AI Chatbots","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Andrea Mock, Ashish Mehta, Desmond C. Ong, Eric Lin, Jacy Reese Anthis, Jared Moore, Kevin Klyman, Nick Haber, Percy Liang, Ryan Louie, William Agnew, Yifan Mai","submitted_at":"2026-08-05T16:11:08Z","abstract_excerpt":"Mental health professionals have raised concerns about risks of psychological harm from interaction with large language models (LLMs), including \"delusional spirals\" in which concerning human and LLM behaviors reinforce each other over time. With growing public use of LLM-powered chatbots, there is an urgent need to build evaluations grounded in real-world episodes of psychological harm experienced by users. We developed DelusionEval, an evaluation protocol that tests a model's tendencies to exhibit behaviors linked to promoting user delusions. We prompt each model with 589 unique conversation"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.05004","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2026-08-05T16:11:08Z","cross_cats_sorted":[],"title_canon_sha256":"91f171642a7dc59cb2e0a2652123ce85bfb46fbffd40d9c408de0f2d5fe8f5e8","abstract_canon_sha256":"e41fbaa44d601b6d8c3b2c059246d751ed7a3c6852f2d144480af2bb72ca26c3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-06T01:48:02.193683Z","signature_b64":"Hq/LkFywosGxR/0e/nwafyS8n2iu8GOUpop/KjVso/8p2spNCFLX7WL8WiqIRhCugLm1G12PtHzMGVxH513LCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5eed8b33f7e5fef946f656aba57c346037703961397fe9dae99fac70c2cb5ff8","last_reissued_at":"2026-08-06T01:48:02.192222Z","signature_status":"signed_v1","first_computed_at":"2026-08-06T01:48:02.192222Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DelusionEval: Measuring Delusion-Linked Behaviors in AI Chatbots","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Andrea Mock, Ashish Mehta, Desmond C. Ong, Eric Lin, Jacy Reese Anthis, Jared Moore, Kevin Klyman, Nick Haber, Percy Liang, Ryan Louie, William Agnew, Yifan Mai","submitted_at":"2026-08-05T16:11:08Z","abstract_excerpt":"Mental health professionals have raised concerns about risks of psychological harm from interaction with large language models (LLMs), including \"delusional spirals\" in which concerning human and LLM behaviors reinforce each other over time. With growing public use of LLM-powered chatbots, there is an urgent need to build evaluations grounded in real-world episodes of psychological harm experienced by users. We developed DelusionEval, an evaluation protocol that tests a model's tendencies to exhibit behaviors linked to promoting user delusions. We prompt each model with 589 unique conversation"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.05004","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.05004/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.05004","created_at":"2026-08-06T01:48:02.194046+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.05004v1","created_at":"2026-08-06T01:48:02.194046+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.05004","created_at":"2026-08-06T01:48:02.194046+00:00"},{"alias_kind":"pith_short_12","alias_value":"L3WYWM7X4X7P","created_at":"2026-08-06T01:48:02.194046+00:00"},{"alias_kind":"pith_short_16","alias_value":"L3WYWM7X4X7PSRXW","created_at":"2026-08-06T01:48:02.194046+00:00"},{"alias_kind":"pith_short_8","alias_value":"L3WYWM7X","created_at":"2026-08-06T01:48:02.194046+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/L3WYWM7X4X7PSRXWK2V2K7BUMA","json":"https://pith.science/pith/L3WYWM7X4X7PSRXWK2V2K7BUMA.json","graph_json":"https://pith.science/api/pith-number/L3WYWM7X4X7PSRXWK2V2K7BUMA/graph.json","events_json":"https://pith.science/api/pith-number/L3WYWM7X4X7PSRXWK2V2K7BUMA/events.json","paper":"https://pith.science/paper/L3WYWM7X"},"agent_actions":{"view_html":"https://pith.science/pith/L3WYWM7X4X7PSRXWK2V2K7BUMA","download_json":"https://pith.science/pith/L3WYWM7X4X7PSRXWK2V2K7BUMA.json","view_paper":"https://pith.science/paper/L3WYWM7X","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.05004&json=true","fetch_graph":"https://pith.science/api/pith-number/L3WYWM7X4X7PSRXWK2V2K7BUMA/graph.json","fetch_events":"https://pith.science/api/pith-number/L3WYWM7X4X7PSRXWK2V2K7BUMA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/L3WYWM7X4X7PSRXWK2V2K7BUMA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/L3WYWM7X4X7PSRXWK2V2K7BUMA/action/storage_attestation","attest_author":"https://pith.science/pith/L3WYWM7X4X7PSRXWK2V2K7BUMA/action/author_attestation","sign_citation":"https://pith.science/pith/L3WYWM7X4X7PSRXWK2V2K7BUMA/action/citation_signature","submit_replication":"https://pith.science/pith/L3WYWM7X4X7PSRXWK2V2K7BUMA/action/replication_record"}},"created_at":"2026-08-06T01:48:02.194046+00:00","updated_at":"2026-08-06T01:48:02.194046+00:00"}