{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:W2O7KMVLP2QEZNM4HOIPHY3E7M","short_pith_number":"pith:W2O7KMVL","schema_version":"1.0","canonical_sha256":"b69df532ab7ea04cb59c3b90f3e364fb36f650d17acae95154539e4e456b8281","source":{"kind":"arxiv","id":"2401.00820","version":2},"attestation_state":"computed","paper":{"title":"A Computational Framework for Behavioral Assessment of LLM Therapists","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.HC"],"primary_cat":"cs.CL","authors_text":"Ashish Sharma, Inna Wanyin Lin, Tim Althoff, Yu Ying Chiu","submitted_at":"2024-01-01T17:32:28Z","abstract_excerpt":"The emergence of large language models (LLMs) like ChatGPT has increased interest in their use as therapists to address mental health challenges and the widespread lack of access to care. However, experts have emphasized the critical need for systematic evaluation of LLM-based mental health interventions to accurately assess their capabilities and limitations. Here, we propose BOLT, a proof-of-concept computational framework to systematically assess the conversational behavior of LLM therapists. We quantitatively measure LLM behavior across 13 psychotherapeutic approaches with in-context learn"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.00820","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-01-01T17:32:28Z","cross_cats_sorted":["cs.HC"],"title_canon_sha256":"94915bf7f37862a75e2b46c25e25f157ab32fd3de821deb8c2faf4f96ffe9865","abstract_canon_sha256":"1ace39b82acfafe5f14ca38ac33f039f3779930d601eee8fc441916b11c2fd91"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:41:30.796019Z","signature_b64":"UIuhQhlLJ9pBSs+9QeHabWOglLQsTHWqhePsAi65HKwdIMwP1Q/ZHNvvKTwXTXIjsYo61eD6ehJfp5VqGndlAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b69df532ab7ea04cb59c3b90f3e364fb36f650d17acae95154539e4e456b8281","last_reissued_at":"2026-07-05T09:41:30.795523Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:41:30.795523Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Computational Framework for Behavioral Assessment of LLM Therapists","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.HC"],"primary_cat":"cs.CL","authors_text":"Ashish Sharma, Inna Wanyin Lin, Tim Althoff, Yu Ying Chiu","submitted_at":"2024-01-01T17:32:28Z","abstract_excerpt":"The emergence of large language models (LLMs) like ChatGPT has increased interest in their use as therapists to address mental health challenges and the widespread lack of access to care. However, experts have emphasized the critical need for systematic evaluation of LLM-based mental health interventions to accurately assess their capabilities and limitations. Here, we propose BOLT, a proof-of-concept computational framework to systematically assess the conversational behavior of LLM therapists. We quantitatively measure LLM behavior across 13 psychotherapeutic approaches with in-context learn"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.00820","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.00820/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.00820","created_at":"2026-07-05T09:41:30.795581+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.00820v2","created_at":"2026-07-05T09:41:30.795581+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.00820","created_at":"2026-07-05T09:41:30.795581+00:00"},{"alias_kind":"pith_short_12","alias_value":"W2O7KMVLP2QE","created_at":"2026-07-05T09:41:30.795581+00:00"},{"alias_kind":"pith_short_16","alias_value":"W2O7KMVLP2QEZNM4","created_at":"2026-07-05T09:41:30.795581+00:00"},{"alias_kind":"pith_short_8","alias_value":"W2O7KMVL","created_at":"2026-07-05T09:41:30.795581+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.21569","citing_title":"When Support Escalates Distress: Regulation and Escalation in LLM Responses to Venting and Advice-Seeking","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2506.09354","citing_title":"\"Is This Really a Human Peer Supporter?\": Misalignments Between Peer Supporters and Experts in LLM-Supported Interactions","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16403","citing_title":"Computational Hermeneutics: Evaluating generative AI as a cultural technology","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02713","citing_title":"Breakdowns in Conversational AI: Interactional Failures in Emotionally and Ethically Sensitive Contexts","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/W2O7KMVLP2QEZNM4HOIPHY3E7M","json":"https://pith.science/pith/W2O7KMVLP2QEZNM4HOIPHY3E7M.json","graph_json":"https://pith.science/api/pith-number/W2O7KMVLP2QEZNM4HOIPHY3E7M/graph.json","events_json":"https://pith.science/api/pith-number/W2O7KMVLP2QEZNM4HOIPHY3E7M/events.json","paper":"https://pith.science/paper/W2O7KMVL"},"agent_actions":{"view_html":"https://pith.science/pith/W2O7KMVLP2QEZNM4HOIPHY3E7M","download_json":"https://pith.science/pith/W2O7KMVLP2QEZNM4HOIPHY3E7M.json","view_paper":"https://pith.science/paper/W2O7KMVL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.00820&json=true","fetch_graph":"https://pith.science/api/pith-number/W2O7KMVLP2QEZNM4HOIPHY3E7M/graph.json","fetch_events":"https://pith.science/api/pith-number/W2O7KMVLP2QEZNM4HOIPHY3E7M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/W2O7KMVLP2QEZNM4HOIPHY3E7M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/W2O7KMVLP2QEZNM4HOIPHY3E7M/action/storage_attestation","attest_author":"https://pith.science/pith/W2O7KMVLP2QEZNM4HOIPHY3E7M/action/author_attestation","sign_citation":"https://pith.science/pith/W2O7KMVLP2QEZNM4HOIPHY3E7M/action/citation_signature","submit_replication":"https://pith.science/pith/W2O7KMVLP2QEZNM4HOIPHY3E7M/action/replication_record"}},"created_at":"2026-07-05T09:41:30.795581+00:00","updated_at":"2026-07-05T09:41:30.795581+00:00"}