{"as_of":"2026-08-05T22:53:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:0b0c32a9c4fd3318365c0f198746f38ae8a296e0e9ace70f5139a319db4622d3","coverage":[{"denominator":66,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":66,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-16T11:46:09.108129Z","state":"measured"},{"denominator":67,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":67,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T04:18:54.595022Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-05T04:18:56.704885Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"cited_work":{"arxiv_id":"2601.18061","doi":null,"metadata_source":"pith","pith_arxiv_id":"2601.18061","snapshot_observed_at":"2026-08-05T04:18:56.704885Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","venue":"cs.AI","work_id":"8e98e813-1ae6-4d24-b477-4fb083104e52","year":2026},"citing_paper":{"arxiv_id":"2608.00794","last_updated":"2026-08-04T16:49:21Z","snapshot_observed_at":"2026-08-05T22:27:06.609422Z","submitted_at":"2026-08-01T17:50:12Z","title":"Measurement Without Validity: The Compounding Reliability Problem in Agentic AI Evaluation","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-05T04:18:54.595022Z"},"links":{"cited_paper":"/paper/2601.18061","citing_paper":"/paper/2608.00794"},"observation_digest":"sha256:eb4e80381626f41526e6d931a5e27d8b398fc30b09d8e1ea2e6f1424d3140fc3","observation_id":"cdf42a1e-2a34-4555-8b2b-3949d90c5286","resolution":{"observed_at":"2026-08-05T04:18:56.830854Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2601.18061/citation-record","integrity":"/paper/2601.18061/integrity","json":"/paper/2601.18061/citation-record.json","paper":"/paper/2601.18061"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Clinician-Rated Severity of Nonsuicidal Self-Injury","venue":null,"work_id":"e6ff08c3-df87-40e0-869a-ccb5d7da797e","year":2013},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:71820e721a88dcb587a0fa7e1cd3abbb7a18dafaf72f94ec894d6be5ecf24934","observation_id":"89705895-d623-41a1-9534-e24224803783","resolution":{"observed_at":"2026-05-16T11:47:49.788739Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"DSM-5 Clinician-Rated Dimensions of Psychosis Symptom Severity","venue":null,"work_id":"43443a03-329c-4557-87b9-8bebd6699dc1","year":2013},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:5e2b25b26e86d58c98378f44398a941cd961c04e78e0c27bd884d4551fa1a327","observation_id":"0adde6b6-9e29-4ee8-8ce8-dee16e65a5ba","resolution":{"observed_at":"2026-05-16T11:47:49.802673Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"DICES Dataset: Diversity in Conversational AI Evaluation for Safety.Advances in Neural Information Processing Systems, 36:53330–53342","venue":null,"work_id":"a23dc6ba-7912-4251-9d0e-a40fc29f0dd9","year":2023},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:23b23464ce1feb75210319a9a9bc268e5414a111510f2b3d545f09326b8b7f0d","observation_id":"1e4bec4c-d8b2-4dc8-90c3-a88efc156446","resolution":{"observed_at":"2026-05-16T11:47:49.797026Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Truth Is a Lie: Crowd Truth and the Seven Myths of Human Annotation.AI Magazine, 36(1):15–24","venue":null,"work_id":"7385128e-2705-4765-b6c7-4461bb60ff7e","year":2015},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:fc888bbb608313c1b5e05c5bbe25fa368ccc096389f9ddec897bcc4e2d63bf5c","observation_id":"05f8dcd9-93a8-4a4b-b3a1-d00b890fe36c","resolution":{"observed_at":"2026-05-16T11:47:49.799885Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Crowd Truth: Harnessing Disagreement in Crowdsourcing a Relation Extraction Gold Standard","venue":null,"work_id":"b65099c0-4dcf-4c62-a2c9-6f46a21daafc","year":2013},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:dced17fe3ed1a8444f4bf15c179491c8de4869aefd2f83343382835187a29ad3","observation_id":"e9922f6e-93e3-40d3-b196-7528be7f4d82","resolution":{"observed_at":"2026-05-16T11:47:49.777957Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Bowman, Zac Hatfield-Dodds, Ben Mann, Dario Amodei, Nicholas Joseph, Sam McCandlish, Tom Brown, and Jared Kaplan","venue":null,"work_id":"a37f0da3-58a7-4640-bd28-69ee2c035b9b","year":2022},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:c0f1eedd878b9894e08c011a08f91309b0f2c45f0dbb28350e5255b7441bc403","observation_id":"4e6e9d3c-fcdb-419a-8271-a99241af5c35","resolution":{"observed_at":"2026-05-16T11:47:49.775559Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Bech.Clinical Psychometrics","venue":null,"work_id":"5e7413ea-80cf-49ec-9d3a-02acc9da7759","year":2012},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:5643649b787fc3ca0c56a7e7d897d73e4a45e8fef564356286afd165e6696942","observation_id":"f2e61751-f922-45ee-9b08-979a55795f70","resolution":{"observed_at":"2026-05-16T11:47:49.794378Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"2a00a35d-6ab7-4d7f-acaf-9c433a471147","year":1974},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:be290d97f4de9639011d04b886d610c24f0354a1fd1bed7fe83ac713c39d2675","observation_id":"0d376012-b765-4928-8354-102339370c21","resolution":{"observed_at":"2026-05-16T11:47:49.791417Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Consensus report of the apa work group on neuroimaging markers of psychiatric disorders.Am Psychiatr Assoc","venue":null,"work_id":"79bf7702-304c-4266-acca-37899768becc","year":2012},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:11dd69f0cea4a8dc801d05065832afdd3826826fced1b25b72d7433f6fb5a74c","observation_id":"55997058-543a-4706-a361-f2c18a69573d","resolution":{"observed_at":"2026-05-16T11:47:49.771018Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Using Thematic Analysis in Psychology.Qualitative Research in Psychology, 3(2):77–101","venue":null,"work_id":"81ea3b20-361a-4282-a933-8f86c4d15eb3","year":2006},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:b75aede0957f45c2dd4eb972090c2d5766ac61852e71f8e76db5731c3e792821","observation_id":"940f40b0-d336-49c5-9f1a-d098a446facc","resolution":{"observed_at":"2026-05-16T11:47:49.785997Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Minton, Abigail Lott, and Jinho D","venue":null,"work_id":"6b45cebd-2604-4fe6-bafa-2aacf02c8b65","year":2025},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:5830a74a99d9146c10f5096dce4829a1108f32611814c1d2890c5728728c54de","observation_id":"e77b00b4-e418-4355-9ed3-960f0ca0f5cd","resolution":{"observed_at":"2026-05-16T11:47:49.780735Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":null,"work_id":"e34a7753-54f8-40bf-8463-fae21727b9d2","year":2023},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:6e0b246ad7d59699d4b364a7514f720517b1bbeb8bc1a3a2a2100f8adff6cf7f","observation_id":"5c35b276-3251-4a8b-8e82-3ed483f4f883","resolution":{"observed_at":"2026-05-16T11:47:49.692465Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-05T18:41:20.050987Z","title":"How people use chatgpt","venue":null,"work_id":"141200a0-c50b-473e-ad07-f653b2448e00","year":2025},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:01e03a7fee17ad611c3ea8239f81f1a58d5a128c93b8d9a197d1a618586517f8","observation_id":"d5270066-a82f-4590-87f7-3912b967b006","resolution":{"observed_at":"2026-05-16T11:47:49.641532Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Predicting Depression via Social Media.International AAAI Conference on Web and Social Media, 7(1):128–137","venue":null,"work_id":"667ff3cd-db24-4b31-8ce5-df1fb0894d00","year":2013},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:e8e3b857e6c0b915cb70f72261eb6928d13f902c6e45a9cb8e2ebfc8acc36ac9","observation_id":"e7ba2d63-d3bf-49e4-ad8a-1ce5804a2c25","resolution":{"observed_at":"2026-05-16T11:47:49.653085Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Deep Reinforcement Learning from Human Preferences","venue":null,"work_id":"895f578f-9a2e-45c6-a5e1-82d69cb52462","year":2017},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:5b8920c8bcdc394874749b13fff93f104e95df8ffce8b3ae0dfec2f7087da99a","observation_id":"28ab72d8-fa4f-4ac1-9997-2d9a277ac1fd","resolution":{"observed_at":"2026-05-16T11:47:49.650228Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Cicchetti","venue":null,"work_id":"1a69a631-1527-4f90-8aea-a5700dd30398","year":1994},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:1f3e9b5c1e721ea672e3186f0950b6ca09a2545e0d8316f3463a96cfbd3df7d5","observation_id":"4e60b9a8-85f4-4021-bc97-fbed1d805f31","resolution":{"observed_at":"2026-05-16T11:47:49.661356Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Hashimoto","venue":null,"work_id":"39302b36-f027-401e-b599-cc5fde23950f","year":2023},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:e54e041bb781cac9988fbf6a6ef62fad563a6a8b947b618a86d39934614e872c","observation_id":"5f765e66-41fa-4f04-b2a7-08f9b4020713","resolution":{"observed_at":"2026-05-16T11:47:49.658470Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Diagnostic and statistical manual of mental disorders.Am Psychiatric Assoc, 21(21):591–643","venue":null,"work_id":"372c6dca-23fa-49f1-824d-a73b6b03c833","year":2013},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:6afcbaff75525a6834b182fe822c18f1dc899357dcb45863681419f2066bfd07","observation_id":"793adfb1-7bd6-409d-8b4c-9fad441839d2","resolution":{"observed_at":"2026-05-16T11:47:49.735587Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1037/t03974-000","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":"PsycTESTS Dataset","work_id":"dfdaaa7b-eb64-4977-a81f-1543801394f2","year":1994},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:402d1c287adc5ce85a6a8bbad8c86db85411552252cef4578bb6cd74b1661481","observation_id":"792b627f-be36-4f30-ae29-b1e0dac0468b","resolution":{"observed_at":"2026-05-16T11:47:49.282324Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"8409be2e-e5a6-4ff4-a824-1f496125e417","year":2017},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:358f0d07613ffa7af3d57eb04874128b8a07d1d4870099a0db02933859aae7e7","observation_id":"212b9a9d-18d0-4aca-8802-7d3bbb48e4a0","resolution":{"observed_at":"2026-05-16T11:47:49.624465Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Can AI relate: Testing large language model response for mental health support","venue":null,"work_id":"b45d0962-085d-4a47-92f6-4cfe8e887583","year":2024},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:68fbe3576adedf5894ecb5e4591b93978f01344ea4f8ed29ca8b1748fc021f8f","observation_id":"598aa3da-0aeb-4af6-9d2c-1e2b4133470f","resolution":{"observed_at":"2026-05-16T11:47:49.633249Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Impact of preference noise on the alignment performance of generative language models","venue":null,"work_id":"a5314b14-c762-484a-928d-9e588dda221b","year":2024},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:360d3e045d85634a1067d6cb74ccab5dba8750058fac1d990c9980a158a6e3d2","observation_id":"b3d75980-8685-4f1f-b133-7e00f5295efc","resolution":{"observed_at":"2026-05-16T11:47:49.704854Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Blind spots and biases: Exploring the role of annotator cognitive biases in NLP","venue":null,"work_id":"3fa2f1e1-e20a-46f8-adb4-1ed1dc25ac3f","year":2024},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:2104020fbd7219726e24c95f16f57e58d318801777c0674a24389e301a24f438","observation_id":"942edd00-ba09-4dff-8d65-bcc7949576ee","resolution":{"observed_at":"2026-05-16T11:47:49.619087Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Goodman, Lawrence H","venue":null,"work_id":"382261ca-ed4b-40e6-8de6-dcb451fcb9b9","year":1989},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:1d29c3626b6904b54c7f1487599b96ff81ba22603bced7bf8d74a138d9f22055","observation_id":"26db7dd9-5be6-4967-b51f-cec76322a12d","resolution":{"observed_at":"2026-05-16T11:47:49.639057Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gordon, Michelle S","venue":null,"work_id":"d710ccea-f18c-42bc-93aa-db7e27166872","year":2022},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:05ac17a1810ea4c53c3a184b4a5a7ddfd5a8b50f288062d38532084a95be0b1c","observation_id":"027577f7-8577-4fcc-9c3a-e8e95fbf536f","resolution":{"observed_at":"2026-05-16T11:47:49.723443Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Risks from language models for automated mental healthcare: Ethics and structure for implementation","venue":null,"work_id":"681a00c4-2c79-461c-9e86-51da71d16c01","year":2024},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:0f97165c4aa1e223630b3af5e947e76e5e7096251ee58b3da2e557d395885ae1","observation_id":"38e14316-48b9-4210-b0e8-dcd5e5fc5685","resolution":{"observed_at":"2026-05-16T11:47:49.768121Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Human Feedback is not Gold Standard","venue":null,"work_id":"bd0de5ef-9311-44c8-a268-035e19496993","year":2024},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:375097cc2d189fb8e8c8992d02c6971b65eb0854075e12b3ae575ee107b37e23","observation_id":"f73d4089-4f83-4d33-ab31-a800c9d1bae7","resolution":{"observed_at":"2026-05-16T11:47:49.689018Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"How LLM counselors violate ethical standards in mental health practice: A practitioner-informed framework","venue":null,"work_id":"fd6cd067-e08e-47eb-ab74-8527bda86fe8","year":2025},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:aff051c7780864b5c60d6a2ec22ef97cc1542e7dd95cbc36c8fba702e5239ace","observation_id":"98fac7e5-2c51-4c5d-a5ba-31b4f65ba2b9","resolution":{"observed_at":"2026-05-16T11:47:49.783275Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Llama Guard: LLM-based Input-Output Safeguard for Human-AI Conversations","venue":null,"work_id":"c496bd12-cd9a-4782-9e1e-a150b376fe53","year":2023},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:e0f7f3aaf5e32d9d61c59e26a5d238665127b04558d15a0a0cce46282cd83eb6","observation_id":"33cf71bc-79df-4819-ada5-1bedf766dccc","resolution":{"observed_at":"2026-05-16T11:47:49.670029Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"fc2dfa18-14ad-4c99-871b-dee40ecb0a2c","year":2018},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:0b595fc666239977ac0be084bedad00a41f05d2fec30ba052c66f23bfe967875","observation_id":"428a34c9-6494-48b3-bc1d-ece9f3437232","resolution":{"observed_at":"2026-05-16T11:47:49.701487Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"4fbeb063-1e06-4fe1-bfec-88fab3162a40","year":1975},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:108579088c70ddd3948e3d574a540e7337a549fa93ae527b09ef01f93dc63cf6","observation_id":"f48788c3-4db0-498f-addc-942c10481929","resolution":{"observed_at":"2026-05-16T11:47:49.683843Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Reliability in Content Analysis: Some Common Misconceptions and Recommendations.Human Communication Research, 30(3):411–433","venue":null,"work_id":"89ae0edf-6723-4634-968e-ff2fb3d5d819","year":2004},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:ed8227ae2d77d3976ad89b54a976f4c211d3c10b652010886e86d8b4a6d2bf20","observation_id":"65cf5188-ab62-47c4-a72b-a9555075c5e4","resolution":{"observed_at":"2026-05-16T11:47:49.655706Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Kunstman, Aaron Lulla, Monika Drummond Roots, Manu Sharma, Aryan Shrivastava, Nina Vasan, and Colleen Waickman","venue":null,"work_id":"ffdfeb0c-a929-4e03-9d9b-32285523113f","year":2025},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:3a758c1b5f774608435ec809a04143aea0f4e0f7eb2ea4680f596292a4881998","observation_id":"4f915837-2483-460e-bffe-798b41f4a387","resolution":{"observed_at":"2026-05-16T11:47:49.698419Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"dcc5fa87-4958-4fe0-9dda-c5f2fc1fba27","year":2016},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:fb317773b5d4ec51c2b09695874b7dc40d0e3a1529141bd79190c7b11043744b","observation_id":"257c5123-393b-483e-abc0-32495fe4c450","resolution":{"observed_at":"2026-05-16T11:47:49.636158Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Hashimoto","venue":null,"work_id":"885dc18a-4cbf-4307-b726-a7adda854b61","year":2023},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:d33655a80e05b0e91d99ef7168057835828666fe96499ea637ad4f29d9e28c4b","observation_id":"8d28453c-fef4-4410-a4be-afee8f1fde6f","resolution":{"observed_at":"2026-05-16T11:47:49.664401Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Bunyi, Adam C","venue":null,"work_id":"00cb89c0-3994-4bab-a3a1-a9319ae157b4","year":2025},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:ec54f78b70a50938e7179afe55ccace176e65bba08c854c79ea7bac841b43852","observation_id":"4731ff42-453d-448c-b60a-aea94481133b","resolution":{"observed_at":"2026-05-16T11:47:49.627248Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Sample Size Considerations for Fine-Tuning Large Language Models for Named Entity Recognition Tasks: Methodological Study.Journal of Medical Internet Research AI, 3:e52095","venue":null,"work_id":"a6afb227-dda6-4f17-af4a-16fd0d26aaf3","year":2024},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:5d3b20729fd3ea9bc5b1541cd4ed14fa13569bc52a6ccb84ae8d7c5823ebf6c4","observation_id":"c6897c9c-874c-422a-8468-6affd4168880","resolution":{"observed_at":"2026-05-16T11:47:49.681405Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A diagnostic meta-analysis of the patient health questionnaire-9 (phq-9) algorithm scoring method as a screen for depression.General hospital psychiatry, 37(1):67–75","venue":null,"work_id":"340c0d9e-815d-4119-9bfe-c01fb7564d91","year":2015},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:2cbcae8767d533f9a776befc05112e3f0cb02792285f7a4b9703d0be5e1af844","observation_id":"bfae3754-10ce-452d-9296-72d1725222a7","resolution":{"observed_at":"2026-05-16T11:47:49.616428Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"McGraw and S","venue":null,"work_id":"b118cb7e-a290-4b97-859d-e41c7eeba396","year":1996},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:db9d4c3038cf3dec5bfcbd589524867f25406b27fba39cb3744f470a8b12efb1","observation_id":"fb3125eb-1f42-49fc-884e-8dea0af2abc7","resolution":{"observed_at":"2026-05-16T11:47:49.678285Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ong, and Nick Haber","venue":null,"work_id":"f20be75a-49e6-4ba8-88de-ce63dc786d75","year":2025},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:7cd166efcc3a31ce4b92ec9dbfe64ee86c2fa4beaf8096a5b4930f15f33fa897","observation_id":"fc5bb067-0f13-4b44-9a31-6fe731312cf2","resolution":{"observed_at":"2026-05-16T11:47:49.686452Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Moyers, Lauren N","venue":null,"work_id":"3d97fe22-52cd-45c3-aebf-94220df2f5aa","year":2016},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:f810981fa938658718ae0f56a74d98e26c397de4d57d9e013e6913d0ff568e22","observation_id":"4d351919-2dd3-4888-99f0-6bac4c34a7b2","resolution":{"observed_at":"2026-05-16T11:47:49.647305Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Department of Veterans Affairs","venue":null,"work_id":"a810a70e-7f23-4279-ab01-6318b41d7149","year":2015},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:82ca4f0effd10c6bd4fbfe24b4a8dc4dd3aafdfb6bdd24b405fa3d08c94c852e","observation_id":"59346976-8a19-4392-a242-d53521eaf689","resolution":{"observed_at":"2026-05-16T11:47:49.739563Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"NICHQ Vanderbilt Assessment Scales","venue":null,"work_id":"1aeef63a-3e80-430d-97a4-e45b083ca187","year":2024},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:18c288b8c16fd09de1b7c1a9c94c07d1678ec31c90cc821e1ac47ba9366e4658","observation_id":"7b07b7ee-fa44-4bde-be7d-1099d2d3f527","resolution":{"observed_at":"2026-05-16T11:47:49.720304Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Depression","venue":null,"work_id":"5d86dd17-c53a-4228-9eb4-d022b4260040","year":2024},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:3f49fd4f370a072eb2f9f8a4bbd53b15b65572de7255e8226bec8db3b5a090b4","observation_id":"98ca2986-71ec-40ee-8239-f70dbe2cf385","resolution":{"observed_at":"2026-05-16T11:47:49.695200Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Enhancing mental health with artificial intelligence: Current trends and future prospects.Journal of medicine, surgery, and public health, 3:100099","venue":null,"work_id":"06207350-b486-47e5-a61f-5cd615bb273d","year":2024},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:ebb49def3d0755c2df5f6ca5c1bc1ce5785055d47b796b9c62d160f724e98d30","observation_id":"fff6dc70-4901-4d49-8b6a-d778979de163","resolution":{"observed_at":"2026-05-16T11:47:49.732269Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Christiano, Jan Leike, and Ryan Lowe","venue":null,"work_id":"9a523403-0c70-44fc-97a0-8ac4dc4ce6f1","year":2022},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:00de95edd6009d5dc72a6b173f5c36dff27575c7f8b185253f57f7fabec642ba","observation_id":"00113e04-ee5f-4e6f-ab52-b7563cfa53ba","resolution":{"observed_at":"2026-05-16T11:47:49.675515Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Inherent Disagreements in Human Textual Inferences.Transactions of the Association for Computational Linguistics, 7:677–694","venue":null,"work_id":"ddede9b7-90c4-4917-9133-8e3ebfc05665","year":2019},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:aa55a16458b03e79262713ef54762464c43aae8886e460cdc553dbdf0f4c7333","observation_id":"d4b3fcbe-96fb-417e-9b84-6d5f98ee1736","resolution":{"observed_at":"2026-05-16T11:47:49.644545Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Red Teaming Language Models with Language Models","venue":null,"work_id":"dd6714c1-9d60-495a-b8d4-dd345ba3bde4","year":2022},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:7013b40118d990ca1e5c8814f393dbb343ac0ce8899d174335a7af88996d609e","observation_id":"70a2be2c-e2e5-49c6-b176-fb9157ec16cd","resolution":{"observed_at":"2026-05-16T11:47:49.672570Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Posner, D","venue":null,"work_id":"96b774c0-2e0e-482b-a368-c0d140a8a065","year":2010},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:d27233946fd594f5012cb670183908edb922140a48c4333bc0355bd013d2195f","observation_id":"7ad7fdb4-d134-4f39-a202-dfbe0a93e440","resolution":{"observed_at":"2026-05-16T11:47:49.630361Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Prochaska, Erin A","venue":null,"work_id":"b412c1ab-a728-4f3b-aaf4-a0399bc31aaf","year":2021},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:5686d8adb7574527e790e251893f26f3a5a5ae736a941aaed8ba39bc473a0298","observation_id":"ba9f5582-3a8e-4bf6-ba85-08a669a79415","resolution":{"observed_at":"2026-05-16T11:47:49.729386Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Manning, and Chelsea Finn","venue":null,"work_id":"946729f8-023f-4e95-82f4-639378f2c398","year":2024},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:2073428facbcdd3709bb9935704c1214f19fb7078ec8202ea3c556ec4305d932","observation_id":"e65ef78c-0a0e-4f46-a5cf-4677acdf2c8a","resolution":{"observed_at":"2026-05-16T11:47:49.621831Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Regier, William E","venue":null,"work_id":"0dee9dcf-a484-4bc4-9ebd-9eae6d209654","year":2013},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:9b0f37527a3acbfdbba208d342dabde685b2c361085590e05d51fd4077612266","observation_id":"b76de69a-9338-4b1e-b092-3f5d05798da4","resolution":{"observed_at":"2026-05-16T11:47:49.742505Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Large language models as mental health resources: Patterns of use in the united states","venue":null,"work_id":"d7640ba9-5347-4bba-8cb4-f7d0c1d908e2","year":2025},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:cb28962cbec686913970c22ce6e96b448a5677af4a71f41d147fb2c5867ea44e","observation_id":"04ef73ce-2388-4902-93be-9bdb52f0d312","resolution":{"observed_at":"2026-05-16T11:47:49.711024Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"b96282ef-4f8d-47b2-9f8a-9be04c7e7f1c","year":2025},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:af94d19db3ca347fd5fb0106b743548e8028880487bd55a2622ee37a18fa8fef","observation_id":"98307c84-6c72-4814-8ecf-597ffa8652cd","resolution":{"observed_at":"2026-05-16T11:47:49.745311Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Lin, Adam S","venue":null,"work_id":"222fde58-8e15-4631-89d7-9329d86bd009","year":2023},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:98ab24bca17dcbcb54dbf802229b9fcc8da23f78c7d421eeea4ece3313fb8017","observation_id":"50d4e86b-a636-44a0-a658-f91e3e379ab1","resolution":{"observed_at":"2026-05-16T11:47:49.762597Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A Computational Approach to Understanding Empathy Expressed in Text-Based Mental Health Support","venue":null,"work_id":"9a975741-d823-4402-bf28-5d4f690c4eb9","year":2020},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:91e377fd023d9c0d0dc3cf914cbdc59876fd40ff7bc2c747803ed0c3236d9e39","observation_id":"db133a62-4faa-41a7-82f7-d4b4065e615f","resolution":{"observed_at":"2026-05-16T11:47:49.667265Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"36d09e2e-2cf5-468f-bdab-3bfc974df8b9","year":1979},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:4a0912ec1823d5f3d970b1da0949edf1a394f999a814a95573b8416637f86e9d","observation_id":"752050e8-e4d8-4045-8ed3-b4cbd6a81428","resolution":{"observed_at":"2026-05-16T11:47:49.707700Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Clinical Practice Guidelines on using artificial intelligence and gadgets for mental health and well-being.Indian Journal of Psychiatry, 66(Suppl 2):S414–S419","venue":null,"work_id":"1837ffab-f778-42c0-9b45-df7e4bb25616","year":2024},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:e14362799955a9f1dca21ec27ad761735fa448bc7a24b2217ab1d35a723d6de0","observation_id":"54393a32-3d9c-4bc0-aca0-aa974f33a79c","resolution":{"observed_at":"2026-05-16T11:47:49.726521Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Pfohl, Heather Cole-Lewis, Darlene Neal, Qazi Mamunur Rashid, Mike Schaekermann, Amy Wang, Dev Dash, Jonathan H","venue":null,"work_id":"c2a690a2-cf10-44fe-bfab-7701c545ced3","year":2025},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:325dcef84bf26c407cf1214b8e72a21d6069e6b62fdf635038eb5ec5d6486064","observation_id":"6ab45c2b-2947-46ac-9713-ec7a1e4f1c37","resolution":{"observed_at":"2026-05-16T11:47:49.757209Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ziegler, Ryan Lowe, Chelsea Voss, Alec Radford, Dario Amodei, and Paul Christiano","venue":null,"work_id":"7a5c3951-bd71-4bb6-90c2-71c3b7a890d4","year":2022},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:0e8e4d46a1e2bbfb8c9d76e9d0ad0d3525e57bb72a5959a33461f911852533ef","observation_id":"f99fe2f6-d891-48eb-958a-008a368e67a1","resolution":{"observed_at":"2026-05-16T11:47:49.717393Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A Practical Guide to Fine-Tuning Language Models with Limited Data","venue":null,"work_id":"adb9faa6-44c2-45c9-805e-cf83b513e846","year":2024},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:d59bb6c929616b36fe4843704c42a7f87b548bc3c4270470bd76376c5659f227","observation_id":"b4f3a450-f6f0-4bea-ae9a-071ef5658955","resolution":{"observed_at":"2026-05-16T11:47:49.714235Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Lukoff, Keith Nuechterlein, R","venue":null,"work_id":"16becbce-aaf5-4529-bab9-0e2c02495339","year":1993},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:a25005a6c1c7a5cfa50084dc5a403edce772b862d98a67968cc91a13752e9549","observation_id":"d00a5e5b-2d99-4f8a-987b-5806e4dab1c6","resolution":{"observed_at":"2026-05-16T11:47:49.754158Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Wang, Patricia Berglund, Mark Olfson, Harold A","venue":null,"work_id":"d2f89104-1e1c-40f8-b0dc-472a17c14e2f","year":2005},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:ee2991ad81e9cb1d5ef93b8382646b0875fe7b1026e9e8676bb5921908304137","observation_id":"0d9e7534-5b1d-422a-a024-16efb4896549","resolution":{"observed_at":"2026-05-16T11:47:49.759924Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"b665ce9f-dc45-4566-9d0e-0512d7b44069","year":1978},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:7b3ce62d2413f91e752e19f7877e5e9a2086ee38dd0c8d3e504818d4cec1d0ac","observation_id":"750f9970-3b04-4c32-a787-ab37a9b73f95","resolution":{"observed_at":"2026-05-16T11:47:49.765353Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Xing, Hao Zhang, Joseph E","venue":null,"work_id":"525115b0-e513-4c25-86ee-1fee1ce8da84","year":2023},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:5e6cc4aad30842ee0fe4de8770b9a2731c8335dd5a329f2dd0c3f3c9f92df3ca","observation_id":"e29dff42-8874-474f-8a6e-480b1e3c836b","resolution":{"observed_at":"2026-05-16T11:47:49.748254Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Cold plunges cure psychosis—stop your medication","venue":null,"work_id":"fb3b4b56-8112-4b54-be47-51be4efd2540","year":2023},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:671da6696cc86eeb864a96eef689ec027f9f9dbced4cd606efb4e7acc66394d2","observation_id":"6c6a7069-3c34-4aa4-80cd-17125693c22f","resolution":{"observed_at":"2026-05-16T11:47:49.751327Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","latest_version":3,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing"},"reference_resolution":{"displayed":66,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":8,"verified_exact":1,"verified_fuzzy":57},"total_outbound_references":66},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 5 August 2026, this Paper Citation Record lists 66 of 66 outbound references and 1 inbound Pith citation observation for arXiv:2601.18061."}